diff --git a/CHANGELOG.md b/CHANGELOG.md index f15c72dc..f2f05bb6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,6 +11,9 @@ ### Fixed - (`core`) Fix a RuntimeException error when automatically correcting deprecated data paths in task args +### Changed +- (`sklearn`) Add `max_cores` to the sklearn estimators to limit the number of CPU cores pre-allocated for the training. + ## 11.0.1.0 - 2026-07-02 ### Added diff --git a/khiops/sklearn/estimators.py b/khiops/sklearn/estimators.py index d2b28d94..e2daac2f 100644 --- a/khiops/sklearn/estimators.py +++ b/khiops/sklearn/estimators.py @@ -235,6 +235,8 @@ class KhiopsEstimator(ABC, BaseEstimator): Parameters ---------- + n_cores : int, optional + Maximum number of CPU cores allocated and used for the training. verbose : bool, default `False` If `True` it prints debug information and it does not erase temporary files when fitting, predicting or transforming. @@ -248,12 +250,14 @@ class KhiopsEstimator(ABC, BaseEstimator): def __init__( self, + n_cores=None, verbose=False, output_dir=None, auto_sort=True, ): # Set the estimator parameters and internal variables self._khiops_model_prefix = None + self.n_cores = n_cores self.output_dir = output_dir self.verbose = verbose self.auto_sort = auto_sort @@ -351,6 +355,15 @@ def _cleanup_computation_dir(self, computation_dir): def fit(self, X, y=None, **kwargs): """Fit the estimator + Parameters + ---------- + X : :external:term:`array-like` of shape (n_samples, n_features_in) or dict + Training dataset. Either an :external:term:`array-like` or a ``dict`` + specification for multi-table datasets (see :doc:`/multi_table_primer`). + + y : :external:term:`array-like` of shape (n_samples,) + The target values. + Returns ------- self : KhiopsEstimator @@ -637,6 +650,8 @@ class KhiopsCoclustering(ClusterMixin, KhiopsEstimator): to speed up the processing. This affects the [predict][] method. *Note* The sort by key is performed in a left-to-right, hierarchical, lexicographic manner. + n_cores : int, optional + Maximum number of CPU cores allocated and used for the training. Attributes ---------- @@ -664,11 +679,13 @@ def __init__( build_name_var=True, build_distance_vars=False, build_frequency_vars=False, + n_cores=None, ): super().__init__( verbose=verbose, output_dir=output_dir, auto_sort=auto_sort, + n_cores=n_cores, ) self._khiops_model_prefix = "CC_" self.build_name_var = build_name_var @@ -770,6 +787,7 @@ def _fit_train_model(self, ds, computation_dir, **kwargs): main_table_path, variables, coclustering_file_path, + max_cores=self.n_cores, log_file_path=train_log_file_path, trace=self.verbose, ) @@ -1186,7 +1204,22 @@ def _transform_prepare_deployment_for_predict(self, _): return self.model_.copy(), None def fit_predict(self, X, y=None, **kwargs): - """Performs clustering on X and returns result (instead of labels)""" + """Performs clustering on X and returns result (instead of labels) + + Parameters + ---------- + X : :external:term:`array-like` of shape (n_samples, n_features_in) or dict + Training dataset. Either an :external:term:`array-like` or a ``dict`` + specification for multi-table datasets (see :doc:`/multi_table_primer`). + + y : :external:term:`array-like` of shape (n_samples,) + The target values. + + Returns + ------- + results : `numpy.array` + """ + return self.fit(X, y, **kwargs).predict(X) @@ -1207,11 +1240,13 @@ def __init__( verbose=False, output_dir=None, auto_sort=True, + n_cores=None, ): super().__init__( verbose=verbose, output_dir=output_dir, auto_sort=auto_sort, + n_cores=n_cores, ) self.n_features = n_features self.n_trees = n_trees @@ -1382,7 +1417,8 @@ def _fit_prepare_training_function_inputs(self, ds, computation_dir): report_file_path, ] - # Build the optional parameters from a copy of the estimator parameters + # Build the optional parameters from a copy + # of the estimator initializer parameters kwargs = self.get_params() # Remove non core.api params @@ -1403,6 +1439,7 @@ def _fit_prepare_training_function_inputs(self, ds, computation_dir): kwargs["max_text_features"] = kwargs.pop("n_text_features") kwargs["text_features"] = kwargs.pop("type_text_features") kwargs["max_parts"] = kwargs.pop("n_feature_parts") + kwargs["max_cores"] = kwargs.pop("n_cores") # Add the additional_data_tables parameter kwargs["additional_data_tables"] = additional_data_tables @@ -1537,6 +1574,7 @@ def __init__( verbose=False, output_dir=None, auto_sort=True, + n_cores=None, ): super().__init__( n_features=n_features, @@ -1551,6 +1589,7 @@ def __init__( verbose=verbose, output_dir=output_dir, auto_sort=auto_sort, + n_cores=n_cores, ) # Data to be specified by inherited classes self._predicted_target_meta_data_tag = None @@ -1731,6 +1770,8 @@ class KhiopsClassifier(ClassifierMixin, KhiopsPredictor): affects the [fit][], [predict][] and [predict_proba][] methods. *Note* The sort by key is performed in a left-to-right, hierarchical, lexicographic manner. + n_cores : int, optional + Maximum number of CPU cores allocated and used for the training. Attributes ---------- @@ -1779,6 +1820,7 @@ def __init__( verbose=False, output_dir=None, auto_sort=True, + n_cores=None, ): super().__init__( n_features=n_features, @@ -1793,6 +1835,7 @@ def __init__( verbose=verbose, output_dir=output_dir, auto_sort=auto_sort, + n_cores=n_cores, ) self.n_pairs = n_pairs self.specific_pairs = specific_pairs @@ -2138,6 +2181,8 @@ class KhiopsRegressor(RegressorMixin, KhiopsPredictor): affects the [fit][] and [predict][] methods. *Note* The sort by key is performed in a left-to-right, hierarchical, lexicographic manner. + n_cores : int, optional + Maximum number of CPU cores allocated and used for the training. Attributes ---------- @@ -2173,6 +2218,7 @@ def __init__( verbose=False, output_dir=None, auto_sort=True, + n_cores=None, ): super().__init__( n_features=n_features, @@ -2187,6 +2233,7 @@ def __init__( verbose=verbose, output_dir=output_dir, auto_sort=auto_sort, + n_cores=n_cores, ) self._khiops_model_prefix = "SNB_" self._khiops_baseline_model_prefix = "B_" @@ -2390,6 +2437,8 @@ class KhiopsEncoder(TransformerMixin, KhiopsSupervisedEstimator): affects the [fit][] and [transform][] methods. *Note* The sort by key is performed in a left-to-right, hierarchical, lexicographic manner. + n_cores : int, optional + Maximum number of CPU cores allocated and used for the training. Attributes ---------- @@ -2431,6 +2480,7 @@ def __init__( verbose=False, output_dir=None, auto_sort=True, + n_cores=None, ): super().__init__( n_features=n_features, @@ -2442,6 +2492,7 @@ def __init__( verbose=verbose, output_dir=output_dir, auto_sort=auto_sort, + n_cores=n_cores, ) self.n_pairs = n_pairs self.specific_pairs = specific_pairs diff --git a/tests/test_sklearn.py b/tests/test_sklearn.py index 9eb2f117..1a1c8b33 100644 --- a/tests/test_sklearn.py +++ b/tests/test_sklearn.py @@ -709,7 +709,8 @@ def setUpClass(cls): ("khiops.core", "train_coclustering"): { "log_file_path": os.path.join( cls.output_dir, "khiops_train_cc.log" - ) + ), + "max_cores": 62, }, ("khiops.core", "simplify_coclustering"): { "max_part_numbers": {"SampleId": 2}, @@ -767,6 +768,7 @@ def setUpClass(cls): "group_target_value": False, "additional_data_tables": {}, "keep_selected_variables_only": False, + "max_cores": 63, } }, "predict": { @@ -797,6 +799,7 @@ def setUpClass(cls): "max_parts": 5, "keep_selected_variables_only": False, "additional_data_tables": {}, + "max_cores": 65, } }, "predict": { @@ -834,6 +837,7 @@ def setUpClass(cls): "numerical_recoding_method": "part Id", "pairs_recoding_method": "part Id", "additional_data_tables": {}, + "max_cores": 67, } }, "predict": { @@ -1419,14 +1423,29 @@ def _test_template( self.expected_kwargs.get(schema_type), source_type ) ) + # the original object is a generator not a list + expected_kwargs_list = list(expected_kwargs_list) + # add the custom_kwargs parameters to the expected ones + if ( + len(expected_kwargs_list) + and custom_kwargs is not None + and len(custom_kwargs) + ): + expected_kwargs_list[0].update(custom_kwargs) + special_kwarg_checkers = ( self.special_kwarg_checkers.get(estimator_type_key) .get(estimator_method) .get((module_name, function_name)) ) + union_of_initializer_params_and_method_params = kwargs + if custom_kwargs is not None and len(custom_kwargs): + union_of_initializer_params_and_method_params.update( + custom_kwargs + ) for expected_kwargs in expected_kwargs_list: self._check_kwargs( - kwargs, + union_of_initializer_params_and_method_params, expected_kwargs=expected_kwargs, special_checkers=special_kwarg_checkers, ) @@ -1452,6 +1471,7 @@ def test_parameter_transfer_classifier_fit_from_monotable_dataframe(self): "n_feature_parts": 3, "group_target_value": False, "keep_selected_variables_only": False, + "n_cores": 63, }, ) @@ -1478,6 +1498,7 @@ def test_parameter_transfer_classifier_fit_from_monotable_dataframe_with_df_y( "n_feature_parts": 3, "group_target_value": False, "keep_selected_variables_only": False, + "n_cores": 63, }, ) @@ -1545,6 +1566,7 @@ def test_parameter_transfer_encoder_fit_from_monotable_dataframe(self): "transform_type_categorical": "part_id", "transform_type_numerical": "part_id", "transform_type_pairs": "part_id", + "n_cores": 67, }, ) @@ -1573,6 +1595,7 @@ def test_parameter_transfer_encoder_fit_from_monotable_dataframe_with_df_y( "transform_type_categorical": "part_id", "transform_type_numerical": "part_id", "transform_type_pairs": "part_id", + "n_cores": 67, }, ) @@ -1636,6 +1659,7 @@ def test_parameter_transfer_regressor_fit_from_monotable_dataframe(self): "construction_rules": ["TableMode", "TableSelection"], "n_feature_parts": 5, "keep_selected_variables_only": False, + "n_cores": 65, }, ) @@ -1657,6 +1681,7 @@ def test_parameter_transfer_regressor_fit_from_monotable_dataframe_with_df_y( "construction_rules": ["TableMode", "TableSelection"], "n_feature_parts": 5, "keep_selected_variables_only": False, + "n_cores": 65, }, ) @@ -1704,6 +1729,9 @@ def test_parameter_transfer_coclustering_fit_from_dataframe(self): estimator_method="fit", schema_type="not_applicable", source_type="dataframe", + extra_estimator_kwargs={ + "n_cores": 62, + }, custom_kwargs={ "fit": { "columns": ("SampleId", "Pos", "Char"),