{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/auroc","entry":"auroc","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":16,"n_papers_ran":6,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":13,"n_samples_ran":6,"n_samples_fingerprinted":3,"n_places":16,"n_places_pointer_only":2,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":1,"ran":5,"unverified":7},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.15180","paper":"/paper/arxiv-2609-15180","title":"RETHINKING CORRECTNESS FOR UNCERTAINTY ESTIMATION IN CLINICAL PREDICTION WITH VISION-LANGUAGE MODELS","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"JasonZuu/EHR-Correctness","path":"core/metrics.py","file_url":"https://github.com/JasonZuu/EHR-Correctness/blob/HEAD/core/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6fec5e3cea8accb9","mcp_get_code":{"code_sha256":"6fec5e3cea8accb9"}},{"arxiv_id":"2608.18116","paper":"/paper/arxiv-2608-18116","title":"You Are What You Prompt: Prompt Quality, Domain Shift, and Uncertainty in Agrifood Vision-Language Models Calidad de prompts, cambio de dominio e incertidumbre en modelos visión-lenguaje agroalimentarios","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"ugritai/pid_agrifood","path":"analysis/pid.py","file_url":"https://github.com/ugritai/pid_agrifood/blob/HEAD/analysis/pid.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"CC0-1.0","inline_ok":true,"code_sha256_prefix":"f638c04aac72f2f2","mcp_get_code":{"code_sha256":"f638c04aac72f2f2"}},{"arxiv_id":"2608.16190","paper":"/paper/arxiv-2608-16190","title":"Decorrelation Is Not Complementarity: Skill, Not Lineage, Governs Trusted-Monitor Ensembles","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"anik-jha/challenger-panels","path":"src/panel.py","file_url":"https://github.com/anik-jha/challenger-panels/blob/HEAD/src/panel.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d628c421ee11951a","mcp_get_code":{"code_sha256":"d628c421ee11951a"}},{"arxiv_id":"2608.03967","paper":"/paper/arxiv-2608-03967","title":"Information-Geometric Forward Policy Training in GFlowNets","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"rodsveiga/infogeometric_gflows","path":"gflows/dag.py","file_url":"https://github.com/rodsveiga/infogeometric_gflows/blob/HEAD/gflows/dag.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4fa349ab4ee4b826","mcp_get_code":{"code_sha256":"4fa349ab4ee4b826"}},{"arxiv_id":"2607.29592","paper":"/paper/arxiv-2607-29592","title":"TOOD: Task-Aware Out-of-Distribution Score Calibration for Continual Learners","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"mostafaelaraby/tood-continual-ood","path":"analysis/toy_confidence_gap.py","file_url":"https://github.com/mostafaelaraby/tood-continual-ood/blob/HEAD/analysis/toy_confidence_gap.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"798ae69ccfa89a8a","mcp_get_code":{"code_sha256":"798ae69ccfa89a8a"}},{"arxiv_id":"2605.08863","paper":"/paper/arxiv-2605-08863","title":"Max-pooling Network Revisited: Analyzing the Role of Semantic Probability in Multiple Instance Learning for Hallucination Detection","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"FUJI1229/Hallucination_Detection","path":"gen/generation/uncertainty/utils/eval_utils.py","file_url":"https://github.com/FUJI1229/Hallucination_Detection/blob/HEAD/gen/generation/uncertainty/utils/eval_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7454cc265b181dec","mcp_get_code":{"code_sha256":"7454cc265b181dec"}},{"arxiv_id":"2508.20718","paper":"/paper/arxiv-2508-20718","title":"Addressing Tokenization Inconsistency in Steganography and Watermarking Based on Large Language Models","date":null,"month_inferred_from_arxiv_id":"2025-08","title_source":"syntology","repo":"ryehr/Consistency","path":"src/tokenization_consistency/metrics.py","file_url":"https://github.com/ryehr/Consistency/blob/HEAD/src/tokenization_consistency/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b4056ca2c7273447","mcp_get_code":{"code_sha256":"b4056ca2c7273447"}},{"arxiv_id":"2410.13831","paper":"/paper/the-disparate-benefits-of-deep-ensembles","title":"The Disparate Benefits of Deep Ensembles","date":"2024-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ml-jku/disparate-benefits","path":"source/utils/metrics.py","file_url":"https://github.com/ml-jku/disparate-benefits/blob/HEAD/source/utils/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"442722aa37fb0ef0","mcp_get_code":{"code_sha256":"442722aa37fb0ef0"}},{"arxiv_id":"2406.15927","paper":"/paper/semantic-entropy-probes-robust-and-cheap","title":"Semantic Entropy Probes: Robust and Cheap Hallucination Detection in LLMs","date":"2024-06-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"oatml/semantic-entropy-probes","path":"semantic_uncertainty/uncertainty/utils/eval_utils.py","file_url":"https://github.com/oatml/semantic-entropy-probes/blob/HEAD/semantic_uncertainty/uncertainty/utils/eval_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ccde4062eb35a7ad","mcp_get_code":{"code_sha256":"ccde4062eb35a7ad"}},{"arxiv_id":"2301.07597","paper":"/paper/how-close-is-chatgpt-to-human-experts","title":"How Close is ChatGPT to Human Experts? Comparison Corpus, Evaluation, and Detection","date":"2023-01-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lsc-1/pecola","path":"lib/metrics.py","file_url":"https://github.com/lsc-1/pecola/blob/HEAD/lib/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a6359426276fb40b","mcp_get_code":{"code_sha256":"a6359426276fb40b"}},{"arxiv_id":"2206.09387","paper":"/paper/label-and-distribution-discriminative-dual","title":"Dual Representation Learning for Out-of-Distribution Detection","date":"2022-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lawliet-zzl/drl","path":"code/OOD_Metric.py","file_url":"https://github.com/lawliet-zzl/drl/blob/HEAD/code/OOD_Metric.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"647ffb2c86d7808f","mcp_get_code":{"code_sha256":"647ffb2c86d7808f"}},{"arxiv_id":"2108.09976","paper":"/paper/revealing-distributional-vulnerability-of","title":"Revealing the Distributional Vulnerability of Discriminators by Implicit Generators","date":"2021-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lawliet-zzl/fig","path":"code_FIG/OODMeasures.py","file_url":"https://github.com/lawliet-zzl/fig/blob/HEAD/code_FIG/OODMeasures.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"647ffb2c86d7808f","mcp_get_code":{"code_sha256":"647ffb2c86d7808f"}},{"arxiv_id":"2106.01216","paper":"/paper/evidential-turing-processes","title":"Evidential Turing Processes","date":"2021-06-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ituvisionlab/EvidentialCalibration","path":"scores.py","file_url":"https://github.com/ituvisionlab/EvidentialCalibration/blob/HEAD/scores.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f48a597c2c8cc253","mcp_get_code":{"code_sha256":"f48a597c2c8cc253"}},{"arxiv_id":"1905.06873","paper":"/paper/das3h-modeling-student-learning-and","title":"DAS3H: Modeling Student Learning and Forgetting for Optimally Scheduling Distributed Practice of Skills","date":"2019-05-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jilljenn/ktm","path":"dmirt.py","file_url":"https://github.com/jilljenn/ktm/blob/HEAD/dmirt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2923dba19a371ca3","mcp_get_code":{"code_sha256":"2923dba19a371ca3"}},{"arxiv_id":"2022.findings-emnlp.497","paper":null,"title":"arXiv:2022.findings-emnlp.497","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"lancopku/Avg-Avg","path":"lib/metrics.py","file_url":"https://github.com/lancopku/Avg-Avg/blob/HEAD/lib/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a6359426276fb40b","mcp_get_code":{"code_sha256":"a6359426276fb40b"}},{"arxiv_id":"2022.findings-emnlp.47","paper":null,"title":"arXiv:2022.findings-emnlp.47","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"lancopku/DAN","path":"metrics.py","file_url":"https://github.com/lancopku/DAN/blob/HEAD/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a6359426276fb40b","mcp_get_code":{"code_sha256":"a6359426276fb40b"}}]}