From d5ebea9b46ef1bc61443c85c51adfd4c6a390154 Mon Sep 17 00:00:00 2001 From: Uriah Finkel Date: Tue, 25 Aug 2026 10:14:02 +0300 Subject: [PATCH 1/7] Add Interventions Avoided scaling regression tests --- tests/test_interventions_avoided_scaling.py | 118 ++++++++++++++++++++ 1 file changed, 118 insertions(+) create mode 100644 tests/test_interventions_avoided_scaling.py diff --git a/tests/test_interventions_avoided_scaling.py b/tests/test_interventions_avoided_scaling.py new file mode 100644 index 00000000..483ccc9e --- /dev/null +++ b/tests/test_interventions_avoided_scaling.py @@ -0,0 +1,118 @@ +import numpy as np +import polars as pl +import pytest + +from rtichoke import create_decision_curve, plot_decision_curve, prepare_performance_data +from rtichoke.processing.plotly_helper_functions import _create_reference_lines_data + + +PROBS = {"model": np.array([0.9, 0.8, 0.7, 0.6, 0.4, 0.3, 0.2, 0.1])} +REALS = np.array([1, 0, 1, 0, 1, 0, 0, 1]) +THRESHOLDS = [0.25, 0.5, 0.75] +EXPECTED_COUNTS = { + 0.25: (1.0, 1.0), + 0.5: (2.0, 2.0), + 0.75: (3.0, 3.0), +} +# These values match the current R static Interventions Avoided definition. +EXPECTED_IA = {-0.0 + 0.25: -25.0, 0.5: 0.0, 0.75: 25.0} +OLD_BUGGY_IA = {0.25: 12.125, 0.5: 24.75, 0.75: 37.375} + + +def _performance_rows() -> pl.DataFrame: + performance_data = prepare_performance_data(probs=PROBS, reals=REALS, by=0.25) + return performance_data.filter(pl.col("chosen_cutoff").is_in(THRESHOLDS)).sort( + "chosen_cutoff" + ) + + +def test_static_interventions_avoided_matches_r_definition_per_100(): + rows = _performance_rows() + + for row in rows.iter_rows(named=True): + threshold = float(row["chosen_cutoff"]) + tn, fn = EXPECTED_COUNTS[threshold] + assert float(row["true_negatives"]) == pytest.approx(tn) + assert float(row["false_negatives"]) == pytest.approx(fn) + + expected = EXPECTED_IA[threshold] + actual = float(row["net_benefit_interventions_avoided"]) + assert actual == pytest.approx(expected) + assert actual != pytest.approx(OLD_BUGGY_IA[threshold]) + + +def test_static_interventions_avoided_count_and_net_benefit_forms_are_equivalent(): + rows = _performance_rows() + prevalence = float(REALS.mean()) + + for row in rows.iter_rows(named=True): + threshold = float(row["chosen_cutoff"]) + tn = float(row["true_negatives"]) + fn = float(row["false_negatives"]) + n = float(row["n"]) + net_benefit = float(row["net_benefit"]) + net_benefit_all = prevalence - (1 - prevalence) * threshold / (1 - threshold) + + from_counts = 100 * (tn / n - fn / n * (1 - threshold) / threshold) + from_net_benefit = ( + 100 + * (net_benefit - net_benefit_all) + * (1 - threshold) + / threshold + ) + + assert float(row["net_benefit_interventions_avoided"]) == pytest.approx( + from_counts + ) + assert from_counts == pytest.approx(from_net_benefit) + + +def test_interventions_avoided_model_and_references_use_same_per_100_unit(): + rows = _performance_rows() + aj = pl.DataFrame({"reference_group": ["population"], "aj_estimate": [0.5]}) + refs = _create_reference_lines_data( + curve="interventions avoided", + aj_estimates_from_performance_data=aj, + multiple_populations=False, + min_p_threshold=0.25, + max_p_threshold=0.75, + ) + + treat_all = refs.filter(pl.col("reference_group") == "treat_all") + assert np.allclose(treat_all["y"].to_numpy(), 0.0) + + treat_none = refs.filter( + (pl.col("reference_group") == "treat_none") + & pl.col("x").is_in(THRESHOLDS) + ).sort("x") + expected_treat_none = np.array( + [100 * (1 - 0.5 - 0.5 * (1 - threshold) / threshold) for threshold in THRESHOLDS] + ) + np.testing.assert_allclose(treat_none["y"].to_numpy(), expected_treat_none) + + np.testing.assert_allclose( + rows["net_benefit_interventions_avoided"].to_numpy(), + np.array([EXPECTED_IA[threshold] for threshold in THRESHOLDS]), + ) + + +def test_static_interventions_avoided_public_apis_are_unchanged(): + performance_data = prepare_performance_data(probs=PROBS, reals=REALS, by=0.25) + + created = create_decision_curve( + probs=PROBS, + reals=REALS, + decision_type="interventions avoided", + by=0.25, + min_p_threshold=0.25, + max_p_threshold=0.75, + ) + plotted = plot_decision_curve( + performance_data, + decision_type="interventions avoided", + min_p_threshold=0.25, + max_p_threshold=0.75, + ) + + assert created.__class__.__name__ == "Figure" + assert plotted.__class__.__name__ == "Figure" From 44ee2d34d26f61c30e94db46ee85aa6dec0a7c7b Mon Sep 17 00:00:00 2001 From: Uriah Finkel Date: Tue, 25 Aug 2026 10:15:07 +0300 Subject: [PATCH 2/7] Fix Interventions Avoided per-100 scaling --- src/rtichoke/processing/transforms.py | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/src/rtichoke/processing/transforms.py b/src/rtichoke/processing/transforms.py index ba162ac1..f5568ccf 100644 --- a/src/rtichoke/processing/transforms.py +++ b/src/rtichoke/processing/transforms.py @@ -684,10 +684,13 @@ def _turn_cumulative_aj_to_performance_data( .alias("net_benefit"), pl.when(pl.col("stratified_by") == "probability_threshold") .then( - 100 * (pl.col("true_negatives") / pl.col("n")) - - (pl.col("false_negatives") / pl.col("n")) - * (1 - pl.col("chosen_cutoff")) - / pl.col("chosen_cutoff") + 100 + * ( + (pl.col("true_negatives") / pl.col("n")) + - (pl.col("false_negatives") / pl.col("n")) + * (1 - pl.col("chosen_cutoff")) + / pl.col("chosen_cutoff") + ) ) .otherwise(None) .alias("net_benefit_interventions_avoided"), From 0098a0c31d4cb0f29e44c7e6e3ae098e35bb14a3 Mon Sep 17 00:00:00 2001 From: Uriah Finkel Date: Tue, 25 Aug 2026 10:15:23 +0300 Subject: [PATCH 3/7] Document Interventions Avoided scaling fix --- CHANGELOG.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index aee483c6..13252445 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,8 @@ +- Fixed Interventions Avoided to apply the per-100 scaling to the full model expression, including the false-negative penalty term. + ## v0.1.36 (21/08/2026) - Fixed several binary and time-dependent curve consistency issues, including cutoff-grid endpoints, binary cutoff equality, time-dependent reference prevalence, gains perfect-reference behavior, custom colors, and plot sizing. From fa62ff7614e909f5482020507f8755a5bbad5859 Mon Sep 17 00:00:00 2001 From: Uriah Finkel Date: Tue, 25 Aug 2026 10:15:50 +0300 Subject: [PATCH 4/7] Tighten Interventions Avoided regression fixtures --- tests/test_interventions_avoided_scaling.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/tests/test_interventions_avoided_scaling.py b/tests/test_interventions_avoided_scaling.py index 483ccc9e..78de0a2b 100644 --- a/tests/test_interventions_avoided_scaling.py +++ b/tests/test_interventions_avoided_scaling.py @@ -15,7 +15,7 @@ 0.75: (3.0, 3.0), } # These values match the current R static Interventions Avoided definition. -EXPECTED_IA = {-0.0 + 0.25: -25.0, 0.5: 0.0, 0.75: 25.0} +EXPECTED_IA = {0.25: -25.0, 0.5: 0.0, 0.75: 25.0} OLD_BUGGY_IA = {0.25: 12.125, 0.5: 24.75, 0.75: 37.375} @@ -86,7 +86,10 @@ def test_interventions_avoided_model_and_references_use_same_per_100_unit(): & pl.col("x").is_in(THRESHOLDS) ).sort("x") expected_treat_none = np.array( - [100 * (1 - 0.5 - 0.5 * (1 - threshold) / threshold) for threshold in THRESHOLDS] + [ + 100 * (1 - 0.5 - 0.5 * (1 - threshold) / threshold) + for threshold in THRESHOLDS + ] ) np.testing.assert_allclose(treat_none["y"].to_numpy(), expected_treat_none) From ea74b3f5da35c2c6de12c91f64504c0c424899f1 Mon Sep 17 00:00:00 2001 From: Uriah Finkel Date: Tue, 25 Aug 2026 10:17:07 +0300 Subject: [PATCH 5/7] Format Interventions Avoided regression tests --- tests/test_interventions_avoided_scaling.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/tests/test_interventions_avoided_scaling.py b/tests/test_interventions_avoided_scaling.py index 78de0a2b..05c7ea9f 100644 --- a/tests/test_interventions_avoided_scaling.py +++ b/tests/test_interventions_avoided_scaling.py @@ -2,7 +2,11 @@ import polars as pl import pytest -from rtichoke import create_decision_curve, plot_decision_curve, prepare_performance_data +from rtichoke import ( + create_decision_curve, + plot_decision_curve, + prepare_performance_data, +) from rtichoke.processing.plotly_helper_functions import _create_reference_lines_data From f3ee9a719b5e4bf62d7a6834ce33e391c4b1bd11 Mon Sep 17 00:00:00 2001 From: Uriah Finkel Date: Tue, 25 Aug 2026 10:19:37 +0300 Subject: [PATCH 6/7] Match Ruff formatting for IA tests --- tests/test_interventions_avoided_scaling.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/tests/test_interventions_avoided_scaling.py b/tests/test_interventions_avoided_scaling.py index 05c7ea9f..8f41b189 100644 --- a/tests/test_interventions_avoided_scaling.py +++ b/tests/test_interventions_avoided_scaling.py @@ -86,8 +86,7 @@ def test_interventions_avoided_model_and_references_use_same_per_100_unit(): assert np.allclose(treat_all["y"].to_numpy(), 0.0) treat_none = refs.filter( - (pl.col("reference_group") == "treat_none") - & pl.col("x").is_in(THRESHOLDS) + (pl.col("reference_group") == "treat_none") & pl.col("x").is_in(THRESHOLDS) ).sort("x") expected_treat_none = np.array( [ From 5d96ff3c3ac70c58c33734e3457a380c2de1e6f5 Mon Sep 17 00:00:00 2001 From: Uriah Finkel Date: Tue, 25 Aug 2026 10:20:58 +0300 Subject: [PATCH 7/7] Simplify IA regression test formatting --- tests/test_interventions_avoided_scaling.py | 76 +++++++-------------- 1 file changed, 26 insertions(+), 50 deletions(-) diff --git a/tests/test_interventions_avoided_scaling.py b/tests/test_interventions_avoided_scaling.py index 8f41b189..822c7d7a 100644 --- a/tests/test_interventions_avoided_scaling.py +++ b/tests/test_interventions_avoided_scaling.py @@ -13,39 +13,30 @@ PROBS = {"model": np.array([0.9, 0.8, 0.7, 0.6, 0.4, 0.3, 0.2, 0.1])} REALS = np.array([1, 0, 1, 0, 1, 0, 0, 1]) THRESHOLDS = [0.25, 0.5, 0.75] -EXPECTED_COUNTS = { - 0.25: (1.0, 1.0), - 0.5: (2.0, 2.0), - 0.75: (3.0, 3.0), -} -# These values match the current R static Interventions Avoided definition. -EXPECTED_IA = {0.25: -25.0, 0.5: 0.0, 0.75: 25.0} -OLD_BUGGY_IA = {0.25: 12.125, 0.5: 24.75, 0.75: 37.375} +EXPECTED_IA = [-25.0, 0.0, 25.0] +OLD_BUGGY_IA = [12.125, 24.75, 37.375] +EXPECTED_TN = [1.0, 2.0, 3.0] +EXPECTED_FN = [1.0, 2.0, 3.0] def _performance_rows() -> pl.DataFrame: - performance_data = prepare_performance_data(probs=PROBS, reals=REALS, by=0.25) - return performance_data.filter(pl.col("chosen_cutoff").is_in(THRESHOLDS)).sort( - "chosen_cutoff" - ) + data = prepare_performance_data(probs=PROBS, reals=REALS, by=0.25) + data = data.filter(pl.col("chosen_cutoff").is_in(THRESHOLDS)) + return data.sort("chosen_cutoff") def test_static_interventions_avoided_matches_r_definition_per_100(): + """Expected values characterize the current R static IA definition.""" rows = _performance_rows() - for row in rows.iter_rows(named=True): - threshold = float(row["chosen_cutoff"]) - tn, fn = EXPECTED_COUNTS[threshold] - assert float(row["true_negatives"]) == pytest.approx(tn) - assert float(row["false_negatives"]) == pytest.approx(fn) - - expected = EXPECTED_IA[threshold] - actual = float(row["net_benefit_interventions_avoided"]) - assert actual == pytest.approx(expected) - assert actual != pytest.approx(OLD_BUGGY_IA[threshold]) + np.testing.assert_allclose(rows["true_negatives"].to_numpy(), EXPECTED_TN) + np.testing.assert_allclose(rows["false_negatives"].to_numpy(), EXPECTED_FN) + actual = rows["net_benefit_interventions_avoided"].to_numpy() + np.testing.assert_allclose(actual, EXPECTED_IA) + assert not np.allclose(actual, OLD_BUGGY_IA) -def test_static_interventions_avoided_count_and_net_benefit_forms_are_equivalent(): +def test_static_interventions_avoided_count_and_nb_forms_are_equivalent(): rows = _performance_rows() prevalence = float(REALS.mean()) @@ -56,22 +47,15 @@ def test_static_interventions_avoided_count_and_net_benefit_forms_are_equivalent n = float(row["n"]) net_benefit = float(row["net_benefit"]) net_benefit_all = prevalence - (1 - prevalence) * threshold / (1 - threshold) - from_counts = 100 * (tn / n - fn / n * (1 - threshold) / threshold) - from_net_benefit = ( - 100 - * (net_benefit - net_benefit_all) - * (1 - threshold) - / threshold - ) + from_nb = 100 * (net_benefit - net_benefit_all) * (1 - threshold) / threshold + actual = float(row["net_benefit_interventions_avoided"]) - assert float(row["net_benefit_interventions_avoided"]) == pytest.approx( - from_counts - ) - assert from_counts == pytest.approx(from_net_benefit) + assert actual == pytest.approx(from_counts) + assert from_counts == pytest.approx(from_nb) -def test_interventions_avoided_model_and_references_use_same_per_100_unit(): +def test_interventions_avoided_model_and_references_use_per_100_units(): rows = _performance_rows() aj = pl.DataFrame({"reference_group": ["population"], "aj_estimate": [0.5]}) refs = _create_reference_lines_data( @@ -85,26 +69,18 @@ def test_interventions_avoided_model_and_references_use_same_per_100_unit(): treat_all = refs.filter(pl.col("reference_group") == "treat_all") assert np.allclose(treat_all["y"].to_numpy(), 0.0) - treat_none = refs.filter( - (pl.col("reference_group") == "treat_none") & pl.col("x").is_in(THRESHOLDS) - ).sort("x") - expected_treat_none = np.array( - [ - 100 * (1 - 0.5 - 0.5 * (1 - threshold) / threshold) - for threshold in THRESHOLDS - ] - ) + is_treat_none = pl.col("reference_group") == "treat_none" + is_test_threshold = pl.col("x").is_in(THRESHOLDS) + treat_none = refs.filter(is_treat_none & is_test_threshold).sort("x") + expected_treat_none = [-100.0, 0.0, 100.0 / 3.0] np.testing.assert_allclose(treat_none["y"].to_numpy(), expected_treat_none) - np.testing.assert_allclose( - rows["net_benefit_interventions_avoided"].to_numpy(), - np.array([EXPECTED_IA[threshold] for threshold in THRESHOLDS]), + rows["net_benefit_interventions_avoided"].to_numpy(), EXPECTED_IA ) def test_static_interventions_avoided_public_apis_are_unchanged(): - performance_data = prepare_performance_data(probs=PROBS, reals=REALS, by=0.25) - + data = prepare_performance_data(probs=PROBS, reals=REALS, by=0.25) created = create_decision_curve( probs=PROBS, reals=REALS, @@ -114,7 +90,7 @@ def test_static_interventions_avoided_public_apis_are_unchanged(): max_p_threshold=0.75, ) plotted = plot_decision_curve( - performance_data, + data, decision_type="interventions avoided", min_p_threshold=0.25, max_p_threshold=0.75,