From d2871a4b4e2b6bc0653c02769e9e152f35a76dc5 Mon Sep 17 00:00:00 2001 From: Andres Rodriguez Date: Thu, 3 Sep 2026 12:09:36 -0700 Subject: [PATCH 1/2] Fix warnings emitted by the application scorecard full suite - HyperParametersTuning: make_scorer(needs_proba=...) was removed in scikit-learn 1.6, so every recall scorer failed and GridSearchCV reported non-finite scores. Use response_method="predict_proba". - Test descriptions: round-trip params through NumpyEncoder so params holding DataFrames (fit_params eval_set) no longer fail JSON serialization, which fell back to the default description. - WeakspotsDiagnosis: copy the column slices before adding the bin column (SettingWithCopyWarning). - ScoreProbabilityAlignment: pass observed=True to groupby (pandas FutureWarning). - lending_club: silence scorecardpy's pandas FutureWarnings at its two call sites; they are third-party and hundreds of lines long in the notebook. Co-Authored-By: Claude Fable 5.1 --- tests/test_test_descriptions.py | 23 ++++++++++++++++++ validmind/ai/test_descriptions.py | 4 +++- .../datasets/credit_risk/lending_club.py | 8 +++++-- .../sklearn/HyperParametersTuning.py | 2 +- .../sklearn/ScoreProbabilityAlignment.py | 2 +- .../sklearn/WeakspotsDiagnosis.py | 24 ++++++++++++------- 6 files changed, 50 insertions(+), 13 deletions(-) diff --git a/tests/test_test_descriptions.py b/tests/test_test_descriptions.py index 97cb3f075..a8c6ae10e 100644 --- a/tests/test_test_descriptions.py +++ b/tests/test_test_descriptions.py @@ -17,6 +17,29 @@ class TestTokenEstimation(unittest.TestCase): """Test token estimation and truncation functions.""" + @patch("validmind.api_client.generate_test_result_description") + def test_generate_description_params_are_json_serializable(self, mock_generate): + """Params holding DataFrames (e.g. fit_params eval_set) must not break the request""" + import json + + import pandas as pd + + from validmind.ai.test_descriptions import generate_description + from validmind.vm_models.result import ResultTable + + mock_generate.return_value = {"content": "ok"} + + generate_description( + test_id="validmind.model_validation.sklearn.HyperParametersTuning", + test_description="desc", + tables=[ResultTable(data=[{"Optimized for": "recall", "recall": 0.9}])], + params={"eval_set": [(pd.DataFrame({"a": [1]}), pd.Series([0]))]}, + ) + + payload = mock_generate.call_args[0][0] + json.dumps(payload) # stdlib encoder, as used by requests + self.assertIn("eval_set", payload["params"]) + def test_estimate_tokens_simple(self): """Test simple character-based token estimation.""" # Test with empty string diff --git a/validmind/ai/test_descriptions.py b/validmind/ai/test_descriptions.py index 1a9ac90b6..568db9b98 100644 --- a/validmind/ai/test_descriptions.py +++ b/validmind/ai/test_descriptions.py @@ -171,7 +171,9 @@ def generate_description( "figures": [figure._get_b64_url() for figure in figures or []], "additional_context": additional_context, "instructions": instructions, - "params": params, + # params can hold anything the test accepts (DataFrames, arrays); + # round-trip so the request body is plain JSON + "params": json.loads(json.dumps(params, cls=NumpyEncoder)), } )["content"] diff --git a/validmind/datasets/credit_risk/lending_club.py b/validmind/datasets/credit_risk/lending_club.py index d3d99f050..835aa28d6 100644 --- a/validmind/datasets/credit_risk/lending_club.py +++ b/validmind/datasets/credit_risk/lending_club.py @@ -322,7 +322,9 @@ def woe_encoding(df: pd.DataFrame, verbose: bool = True) -> pd.DataFrame: print(f"Excluded {target_column} from WoE transformation.") # Apply the WoE transformation - df = sc.woebin_ply(df, bins=bins) + with warnings.catch_warnings(): + warnings.simplefilter("ignore") # scorecardpy uses deprecated pandas APIs + df = sc.woebin_ply(df, bins=bins) if verbose: print("Successfully converted features to WoE values.") @@ -379,7 +381,9 @@ def _woebin(df: pd.DataFrame, verbose: bool = True) -> Dict[str, Any]: print( f"Performing binning with breaks_adj: {breaks_adj}" ) # print the breaks_adj being used - bins = sc.woebin(df, target_column, breaks_list=breaks_adj) + with warnings.catch_warnings(): + warnings.simplefilter("ignore") # scorecardpy uses deprecated pandas APIs + bins = sc.woebin(df, target_column, breaks_list=breaks_adj) except Exception as e: print("Error during binning: ") print(e) diff --git a/validmind/tests/model_validation/sklearn/HyperParametersTuning.py b/validmind/tests/model_validation/sklearn/HyperParametersTuning.py index 1ecd8aa83..625b7d2e6 100644 --- a/validmind/tests/model_validation/sklearn/HyperParametersTuning.py +++ b/validmind/tests/model_validation/sklearn/HyperParametersTuning.py @@ -43,7 +43,7 @@ def _create_scoring_dict(scoring, metrics, threshold): for metric in metrics: if metric == "recall": scoring_dict[metric] = make_scorer( - custom_recall, needs_proba=True, threshold=threshold + custom_recall, response_method="predict_proba", threshold=threshold ) elif metric == "roc_auc": scoring_dict[metric] = "roc_auc" diff --git a/validmind/tests/model_validation/sklearn/ScoreProbabilityAlignment.py b/validmind/tests/model_validation/sklearn/ScoreProbabilityAlignment.py index 144559da1..eec621ad5 100644 --- a/validmind/tests/model_validation/sklearn/ScoreProbabilityAlignment.py +++ b/validmind/tests/model_validation/sklearn/ScoreProbabilityAlignment.py @@ -81,7 +81,7 @@ def ScoreProbabilityAlignment( # Calculate statistics per bin results = [] - for bin_name, group in df.groupby("score_bin"): + for bin_name, group in df.groupby("score_bin", observed=True): bin_stats = { "Score Range": f"{bin_name.left:.0f}-{bin_name.right:.0f}", "Mean Score": group[score_column].mean(), diff --git a/validmind/tests/model_validation/sklearn/WeakspotsDiagnosis.py b/validmind/tests/model_validation/sklearn/WeakspotsDiagnosis.py index c0aaf507d..c05b86724 100644 --- a/validmind/tests/model_validation/sklearn/WeakspotsDiagnosis.py +++ b/validmind/tests/model_validation/sklearn/WeakspotsDiagnosis.py @@ -285,14 +285,22 @@ def WeakspotsDiagnosis( figures = [] passed = True - df_1 = datasets[0]._df[ - feature_columns - + [datasets[0].target_column, datasets[0].prediction_column(model)] - ] - df_2 = datasets[1]._df[ - feature_columns - + [datasets[1].target_column, datasets[1].prediction_column(model)] - ] + df_1 = ( + datasets[0] + ._df[ + feature_columns + + [datasets[0].target_column, datasets[0].prediction_column(model)] + ] + .copy() + ) + df_2 = ( + datasets[1] + ._df[ + feature_columns + + [datasets[1].target_column, datasets[1].prediction_column(model)] + ] + .copy() + ) results_1 = pd.DataFrame() results_2 = pd.DataFrame() for feature in feature_columns: From bf110555da3b2b46d6356c6a2565d54e61298346 Mon Sep 17 00:00:00 2001 From: Andres Rodriguez Date: Thu, 3 Sep 2026 12:12:48 -0700 Subject: [PATCH 2/2] Require scikit-learn >= 1.4 make_scorer(response_method=...) exists from scikit-learn 1.4, the release that deprecated needs_proba. Make the floor explicit so an older environment fails at install time rather than with a TypeError at test time. Co-Authored-By: Claude Fable 5.1 --- pyproject.toml | 6 +++--- uv.lock | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index ce818e0dd..aa344669a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -27,7 +27,7 @@ dependencies = [ "plotly (>=6.0.0)", "polars", "python-dotenv", - "scikit-learn", + "scikit-learn (>=1.4)", "seaborn", "tabulate (>=0.9.0,<0.10.0)", "tiktoken", @@ -41,7 +41,7 @@ dependencies = [ all = [ "torch (>=2.0.0)", "xgboost (>=1.5.2,<3.1)", - "scikit-learn (<1.8)", + "scikit-learn (>=1.4,<1.8)", "transformers (>=4.32.0,<5.0.0)", "pycocoevalcap", "ragas (>=0.2.3,<=0.2.7)", @@ -111,7 +111,7 @@ pytorch = ["torch (>=2.0.0)"] stats = ["scipy", "statsmodels (>=0.14.2,<0.15.0)", "arch (>=7.0.0)"] xgboost = [ "xgboost (>=1.5.2,<3.1)", - "scikit-learn (<1.8)", + "scikit-learn (>=1.4,<1.8)", ] explainability = [ "shap (>=0.46.0)", diff --git a/uv.lock b/uv.lock index 28242e3b4..61da4e5d9 100644 --- a/uv.lock +++ b/uv.lock @@ -11519,9 +11519,9 @@ requires-dist = [ { name = "requests", specifier = ">=2.28.0,<3.0.0" }, { name = "rouge", marker = "extra == 'all'", specifier = ">=1" }, { name = "rouge", marker = "extra == 'nlp'", specifier = ">=1" }, - { name = "scikit-learn" }, - { name = "scikit-learn", marker = "extra == 'all'", specifier = "<1.8" }, - { name = "scikit-learn", marker = "extra == 'xgboost'", specifier = "<1.8" }, + { name = "scikit-learn", specifier = ">=1.4" }, + { name = "scikit-learn", marker = "extra == 'all'", specifier = ">=1.4,<1.8" }, + { name = "scikit-learn", marker = "extra == 'xgboost'", specifier = ">=1.4,<1.8" }, { name = "scipy", marker = "extra == 'all'" }, { name = "scipy", marker = "extra == 'stats'" }, { name = "scorecardpy", marker = "extra == 'all'", specifier = "==0.1.9.6" },