From e5ef195cc3e31aec1d9b42b111f9dc09278a03ed Mon Sep 17 00:00:00 2001 From: Tony Liang Date: Thu, 3 Sep 2026 14:50:06 +0800 Subject: [PATCH 1/3] fix: include select feature for sklearn models --- .../usr/bin/load_classifier_class.py | 74 ++++++++++++------- .../resources/usr/bin/run_random_search_cv.py | 7 +- .../usr/bin/sklearn_select_features.py | 2 +- 3 files changed, 55 insertions(+), 28 deletions(-) diff --git a/modules/sklearn/select_feature/resources/usr/bin/load_classifier_class.py b/modules/sklearn/select_feature/resources/usr/bin/load_classifier_class.py index 08f6fd4..b9d91eb 100644 --- a/modules/sklearn/select_feature/resources/usr/bin/load_classifier_class.py +++ b/modules/sklearn/select_feature/resources/usr/bin/load_classifier_class.py @@ -15,21 +15,21 @@ def load_classifier_class(model_name, random_state=42, probability=True): learning_rate = stats.uniform(0.01, 1.1) max_features = ["sqrt", "log2", 100, 500, 1000, None] # Dict to store relevant information of sklearn classifiers - model_info = { - # Logistic regression has built-in predict proba and coef - "Logit": { + + # For logistic regression, it has built-in predict proba and coef + logit_dict = { "class_path": "sklearn.linear_model.LogisticRegression", "default_params": {"C": 1.0, "penalty": "l2", "solver": "liblinear"}, "params_dist": {"C": C, "penalty": ["l2"], "solver": ["liblinear"]} - }, - # Use SVC with kernel linear and not LinearSVC, since the latter do not have predict_proba - "Linear_SVM": { + } + # Use SVC with kernel linear and not LinearSVC, since the latter do not have predict_proba + linear_svm_dict ={ "class_path": "sklearn.svm.SVC", "default_params": {"C": 1.0, "kernel": "linear", "random_state": random_state, "probability": probability}, "params_dist": {"C": C, "kernel": ["linear"] } - }, - # Decision Tree have risk of overfitting, hence require pruning of trees - "Decision_Tree": { + } + # Decision Tree classifier, risk of overfitting, hence require pruning of trees + decision_tree_dict = { "class_path": "sklearn.tree.DecisionTreeClassifier", "default_params": {"max_depth": 10, "random_state": random_state}, "params_dist": { @@ -38,9 +38,9 @@ def load_classifier_class(model_name, random_state=42, probability=True): "min_samples_leaf": min_sample_leaf, "max_leaf_nodes": [10, 100, 1000, None] } - }, - # RandomForest shuold in general work better than single decision tree - "Random_Forest": { + } + # Random Forest classifier, should in general work better than single decision tree + random_forest_dict = { "class_path": "sklearn.ensemble.RandomForestClassifier", "default_params": {"n_estimators": 10, "max_features": "sqrt", "max_depth": 10, "random_state": random_state, "n_jobs": -1}, "params_dist": { @@ -48,19 +48,10 @@ def load_classifier_class(model_name, random_state=42, probability=True): "max_leaf_nodes": [10, 100, 1000, None], "min_samples_leaf": min_sample_leaf } - }, - # AdaBoost is a simpler boosting algorithm - "AdaBoost": { - "class_path": "sklearn.ensemble.AdaBoostClassifier", - "default_params": {"algorithm": "SAMME", "random_state": random_state}, - "params_dist": { - "n_estimators": n_estimators, - "learning_rate": learning_rate, - "algorithm": ['SAMME', 'SAMME.R'] - } - }, - # GradientBoost should tune large number of estimators with slow learning rate - "GradientBoost": { + } + + # Gradient Boost classifier, should tune large number of estimators with slow learning rate + gradient_boost_dict = { "class_path": "sklearn.ensemble.GradientBoostingClassifier", "default_params": {"loss":'log_loss', "learning_rate":0.1, "n_estimators":10}, "params_dist": { @@ -69,7 +60,38 @@ def load_classifier_class(model_name, random_state=42, probability=True): "min_samples_leaf": min_sample_leaf, "max_features": max_features } - } + } + # MLPClassifier is a neural network classifier + mlp_dict = { + "class_path": "sklearn.neural_network.MLPClassifier", + "default_params": {"hidden_layer_sizes": (64,), "max_iter": 500, + "random_state": random_state}, + "params_dist": {"hidden_layer_sizes": [(32,), (64,), (128,), (64, 32)], + "alpha": stats.loguniform(1e-5, 1e-1)} + } + # Mimic xgboost + hist_gradient_boost_dict = { + "class_path": "sklearn.ensemble.HistGradientBoostingClassifier", + "default_params": {"random_state": random_state}, + "params_dist": {"learning_rate": learning_rate, + "max_leaf_nodes": [15, 31, 63], + "min_samples_leaf": stats.randint(5, 30)} + } + + # ======================== + # Lastly merge all together into a single dictionary + model_info = { + "Logit": logit_dict, + "Linear_SVM": linear_svm_dict, + # Decision Tree have risk of overfitting, hence require pruning of trees + "Decision_Tree": decision_tree_dict , + # RandomForest shuold in general work better than single decision tree + "Random_Forest": random_forest_dict, + # GradientBoost should tune large number of estimators with slow learning rate + "Gradient_Boost": gradient_boost_dict, + # MLPClassifier is a neural network classifier + "MLP": mlp_dict, + "Hist_Gradient_Boost": hist_gradient_boost_dict } # Check if valid name of model was input diff --git a/modules/sklearn/select_feature/resources/usr/bin/run_random_search_cv.py b/modules/sklearn/select_feature/resources/usr/bin/run_random_search_cv.py index 8e94adf..4ec6f2b 100644 --- a/modules/sklearn/select_feature/resources/usr/bin/run_random_search_cv.py +++ b/modules/sklearn/select_feature/resources/usr/bin/run_random_search_cv.py @@ -1,11 +1,16 @@ from sklearn.model_selection import RandomizedSearchCV +from sklearn.preprocessing import StandardScaler +from sklearn.pipeline import make_pipeline + # Run a randomized search cv to find optimal params def run_random_search_cv(clf_instance, X, Y, param_distributions, n_iter=10, random_state=42, n_jobs=-1): # Apply random search cross validation to find optimal hyperparameters + # Make sure to use a pipeline to standardize the data before fitting the model + pipeline = make_pipeline(StandardScaler(), clf_instance) # Default to use all processors to speed up process clf_cv = RandomizedSearchCV( - clf_instance, param_distributions=param_distributions, + pipeline, param_distributions=param_distributions, n_iter=n_iter, random_state=random_state, n_jobs=n_jobs ) diff --git a/modules/sklearn/select_feature/resources/usr/bin/sklearn_select_features.py b/modules/sklearn/select_feature/resources/usr/bin/sklearn_select_features.py index 97aae4b..62ee2ef 100755 --- a/modules/sklearn/select_feature/resources/usr/bin/sklearn_select_features.py +++ b/modules/sklearn/select_feature/resources/usr/bin/sklearn_select_features.py @@ -79,7 +79,7 @@ def main(mu_path, dataset_name, model_name, block_num=0, n_iter=10, random_state print(opt_clf_instance) # Apply scaling and fit final model opt_clf = make_pipeline(StandardScaler(), opt_clf_instance) - opt_clf.fit(X_df, y_df["response"]) + opt_clf.fit(X_df, y_df["response"].ravel()) # Extract the classifier from the pipeline classifier = opt_clf.steps[-1][1] # Adjust this based on your pipeline's step name # Then could either extract their weights or feature importance From 7c99113a31fa66df3698f7cdae701dbdbdf878fa Mon Sep 17 00:00:00 2001 From: Tony Liang Date: Thu, 3 Sep 2026 14:50:39 +0800 Subject: [PATCH 2/3] fix: correct sklearn reduction to list param rather than singleton --- subworkflows/methods/sklearn/main.nf | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/subworkflows/methods/sklearn/main.nf b/subworkflows/methods/sklearn/main.nf index 71a778c..2bc2b3a 100755 --- a/subworkflows/methods/sklearn/main.nf +++ b/subworkflows/methods/sklearn/main.nf @@ -33,7 +33,7 @@ workflow SKLEARN { // Classifier to train for sklearn model_name = Channel.fromList(params.sklearn_classifier_names) // Reduction method (pca or empty) - reduction = Channel.value(params.sklearn_reduction) + reduction = Channel.fromList(params.sklearn_reduction) take: // TODO: rename this data_copy to mae_copy or mu_copy depending on language data_copy // ch of tuple dataset, path of mae/mu data, From 6f0ee70b81d32ae5478ff2e09d78db003ab047f3 Mon Sep 17 00:00:00 2001 From: Tony Liang Date: Thu, 3 Sep 2026 14:51:07 +0800 Subject: [PATCH 3/3] fix: correct naming of parameter opt --- conf/real_data.config | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/conf/real_data.config b/conf/real_data.config index a57a822..6b881b7 100644 --- a/conf/real_data.config +++ b/conf/real_data.config @@ -25,7 +25,7 @@ params { skip_mogonet = false skip_mofa = false // Run feature selection - select_feature = true + selectFeature = true k_fold_number = 5 filter_low_var = "1" // This get casted as bool in both Python and R // number of component