{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/arxiv-2601-10863","title":"Beyond Accuracy: A Stability-Aware Metric for Multi-Horizon Forecasting","arxiv_id":"2601.10863","date":"2026-01-15","proceeding":null,"authors":["Chutian Ma","Grigorii Pomazkin","Giacinto Paolo Saggese","Paul Smith"],"abstract":"Traditional time series forecasting methods optimize for accuracy alone. This objective neglects temporal consistency, in other words, how consistently a model predicts the same future event as the forecast origin changes. We introduce the forecast accuracy and coherence score (forecast AC score for short) for measuring the quality of probabilistic multi-horizon forecasts in a way that accounts for both multi-horizon accuracy and stability. Our score additionally allows user-specified weights to balance accuracy and consistency requirements. As an example application, we implement the score as a differentiable objective function for training seasonal auto-regressive integrated models and evaluate it on the M4 Hourly benchmark dataset. Results demonstrate consistent improvements over traditional maximum likelihood estimation. Regarding stability, the AC-optimized model generated out-of-sample forecasts with 15.8\\% reduced variance over forecasts targeting the same timestamp. In terms of accuracy, the AC-optimized model achieved considerable improvements for medium-to-long-horizon forecasts. While one-step-ahead forecasts exhibited a 3.9\\% increase in MSE, forecasts from horizon three onward experienced improved accuracy, with a peak improvement of approximately 6\\% in MSE at horizons 9-12. These results indicate that our metric successfully trains models to produce more stable and accurate multi-step forecasts in exchange for a relatively small degradation in one-step-ahead performance.","url_abs":"https://arxiv.org/abs/2601.10863","url_pdf":"https://arxiv.org/pdf/2601.10863","source":{"archive":null,"snapshot":"2025-07-28","note":"not in the Papers with Code archive (frozen at the snapshot)","row_kind":"graph","title_abstract_authors_date":"arXiv metadata, CC0 1.0 (https://info.arxiv.org/help/license)"},"code_links":[],"tasks":[],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2601.10863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2601.10863"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"mentioned_in_github":null,"is_official":null,"provenance":"deterministic:regex_extraction","mentioned_in_paper":null,"url":"https://github.com/causify-ai/beyond_accuracy","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"unverified":17},"by_repo_kind":{"found_in_text":{"samples":17,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"13f5f6244538eb5a","entry":"add_parallel_processing_arg","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"helpers/hparser.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/helpers/hparser.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"13f5f6244538eb5a"}},{"code_sha256_prefix":"7eeadbc7739a7082","entry":"add_verbosity_arg","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"helpers/hparser.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/helpers/hparser.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"7eeadbc7739a7082"}},{"code_sha256_prefix":"f5d8c6902d16bf9b","entry":"anti_diagonal","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"analyze_outcomes.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/analyze_outcomes.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"f5d8c6902d16bf9b"}},{"code_sha256_prefix":"bf0a010cbb6a5ae9","entry":"antidiag_std_mean","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"analyze_outcomes.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/analyze_outcomes.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"bf0a010cbb6a5ae9"}},{"code_sha256_prefix":"30cbba892dc3a667","entry":"calculate_min_history_statsmodel","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"train_model.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/train_model.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"30cbba892dc3a667"}},{"code_sha256_prefix":"a0ae389e5cd0b603","entry":"compute_ensembled_energy_distance_tensor","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"forecast_metric_utility.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/forecast_metric_utility.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"a0ae389e5cd0b603"}},{"code_sha256_prefix":"3b70fab06112bdb7","entry":"compute_ensembled_energy_score_tensor","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"forecast_metric_utility.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/forecast_metric_utility.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"3b70fab06112bdb7"}},{"code_sha256_prefix":"ce3e529c4fc5b681","entry":"examine_vertical_stability_for_given_target_time","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"analyze_outcomes.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/analyze_outcomes.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"ce3e529c4fc5b681"}},{"code_sha256_prefix":"9bbe596cf1f25c14","entry":"expand_ar_polynomial_torch","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"differentiable_arima.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/differentiable_arima.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"9bbe596cf1f25c14"}},{"code_sha256_prefix":"1e3010a2d91e14a0","entry":"expand_ma_polynomial_torch","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"differentiable_arima.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/differentiable_arima.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"1e3010a2d91e14a0"}},{"code_sha256_prefix":"1fec0d9994f72e6f","entry":"get_memory_usage","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"helpers/hlogging.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/helpers/hlogging.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"1fec0d9994f72e6f"}},{"code_sha256_prefix":"a3d119b9bae01fae","entry":"get_memory_usage_as_str","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"helpers/hlogging.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/helpers/hlogging.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"a3d119b9bae01fae"}},{"code_sha256_prefix":"819e9ecf573ff55e","entry":"line","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"helpers/hprint.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/helpers/hprint.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"819e9ecf573ff55e"}},{"code_sha256_prefix":"8d360144bbd3560a","entry":"memory_to_str","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"helpers/hlogging.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/helpers/hlogging.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"8d360144bbd3560a"}},{"code_sha256_prefix":"56ae6ecc4f028033","entry":"pprint_pformat","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"helpers/hprint.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/helpers/hprint.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"56ae6ecc4f028033"}},{"code_sha256_prefix":"c694609d4437c243","entry":"rolling_forecast_with_statsmodel_arima","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"train_model.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/train_model.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"c694609d4437c243"}},{"code_sha256_prefix":"d9f609def99aa02b","entry":"torch_convolve","repo":"causify-ai/beyond_accuracy","repo_kind":"found_in_text","path":"differentiable_arima.py","file_url":"https://github.com/causify-ai/beyond_accuracy/blob/HEAD/differentiable_arima.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"d9f609def99aa02b"}}]},"arxiv_metadata":{"licence":"arXiv metadata, CC0 1.0 (https://info.arxiv.org/help/license)","fields":["title","abstract","authors","date"],"primary_category":"cs.LG","source":"arxiv_2026.jsonl"},"syntology_extracted_results":null}