{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/trustworthy-deep-learning-via-proper","title":"Better Uncertainty Calibration via Proper Scores for Classification and Beyond","arxiv_id":"2203.07835","date":"2022-03-15","proceeding":null,"authors":["Sebastian G. Gruber","Florian Buettner"],"abstract":"With model trustworthiness being crucial for sensitive real-world applications, practitioners are putting more and more focus on improving the uncertainty calibration of deep neural networks. Calibration errors are designed to quantify the reliability of probabilistic predictions but their estimators are usually biased and inconsistent. In this work, we introduce the framework of proper calibration errors, which relates every calibration error to a proper score and provides a respective upper bound with optimal estimation properties. This relationship can be used to reliably quantify the model calibration improvement. We theoretically and empirically demonstrate the shortcomings of commonly used estimators compared to our approach. Due to the wide applicability of proper scores, this gives a natural extension of recalibration beyond classification.","url_abs":"https://arxiv.org/abs/2203.07835v4","url_pdf":"https://arxiv.org/pdf/2203.07835v4.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"trustworthy-deep-learning-via-proper","repo_url":"https://github.com/MLO-lab/better_uncertainty_calibration","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"trustworthy-deep-learning-via-proper","repo_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2203.07835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.07835"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/MLO-lab/better_uncertainty_calibration","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","reach":null}],"summary":{"ran_fixture":7,"ran_violates":4,"ran_draft_wrong":1,"ran_honours":5,"unverified":3},"by_repo_kind":{"official":{"samples":20,"ran":17,"repositories":2}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":20,"samples":[{"code_sha256_prefix":"5cb88fe9e94e02fa","entry":"_get_ce","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"5cb88fe9e94e02fa"}},{"code_sha256_prefix":"2a8f482dde16d6b6","entry":"bin","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"2a8f482dde16d6b6"}},{"code_sha256_prefix":"e5f6a1151a9a4358","entry":"difference_mean","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"e5f6a1151a9a4358"}},{"code_sha256_prefix":"8c68d185cdf60b2f","entry":"enough_duplicates","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"8c68d185cdf60b2f"}},{"code_sha256_prefix":"64ff9694bed38352","entry":"equal_mass_bins","repo":"mlo-lab/better_uncertainty_calibration","repo_kind":"official","path":"errors.py","file_url":"https://github.com/mlo-lab/better_uncertainty_calibration/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"64ff9694bed38352"}},{"code_sha256_prefix":"6414bbc08df09a48","entry":"equal_width_bins","repo":"mlo-lab/better_uncertainty_calibration","repo_kind":"official","path":"errors.py","file_url":"https://github.com/mlo-lab/better_uncertainty_calibration/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"6414bbc08df09a48"}},{"code_sha256_prefix":"c57c1af0da2c94a6","entry":"fast_bin","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"c57c1af0da2c94a6"}},{"code_sha256_prefix":"0f3d6c736e79c06a","entry":"get_bin_probs","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"0f3d6c736e79c06a"}},{"code_sha256_prefix":"c8a933ef5fbe6075","entry":"get_discrete_bins","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"c8a933ef5fbe6075"}},{"code_sha256_prefix":"e78ed8a6a8a30bb9","entry":"get_labels_one_hot","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"e78ed8a6a8a30bb9"}},{"code_sha256_prefix":"2f700262ea09042a","entry":"get_top_predictions","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"2f700262ea09042a"}},{"code_sha256_prefix":"c6d37bf5556bca2d","entry":"get_top_probs","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"c6d37bf5556bca2d"}},{"code_sha256_prefix":"eb163b70613783c5","entry":"is_discrete","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"eb163b70613783c5"}},{"code_sha256_prefix":"a94f02b8109c002a","entry":"plugin_ce","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"a94f02b8109c002a"}},{"code_sha256_prefix":"815e55dabb053689","entry":"split","repo":"mlo-lab/better_uncertainty_calibration","repo_kind":"official","path":"errors.py","file_url":"https://github.com/mlo-lab/better_uncertainty_calibration/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"815e55dabb053689"}},{"code_sha256_prefix":"eecd90591e3b3066","entry":"unbiased_l2_ce","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"eecd90591e3b3066"}},{"code_sha256_prefix":"065c035e0ef047af","entry":"unbiased_square_ce","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"065c035e0ef047af"}},{"code_sha256_prefix":"8b2382c2062304bf","entry":"get_binning_ce","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"8b2382c2062304bf"}},{"code_sha256_prefix":"5c8ada24afd93532","entry":"get_calibration_error","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"5c8ada24afd93532"}},{"code_sha256_prefix":"bd678d4acf335009","entry":"lower_bound_scaling_ce","repo":"MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond","repo_kind":"official","path":"errors.py","file_url":"https://github.com/MLO-lab/better_uncertainty_calibration_via_proper_scores_for_classification_and_beyond/blob/HEAD/errors.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"bd678d4acf335009"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}