Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion conf/real_data.config
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,7 @@ params {
skip_mogonet = false
skip_mofa = false
// Run feature selection
select_feature = true
selectFeature = true
k_fold_number = 5
filter_low_var = "1" // This get casted as bool in both Python and R
// number of component
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -15,21 +15,21 @@ def load_classifier_class(model_name, random_state=42, probability=True):
learning_rate = stats.uniform(0.01, 1.1)
max_features = ["sqrt", "log2", 100, 500, 1000, None]
# Dict to store relevant information of sklearn classifiers
model_info = {
# Logistic regression has built-in predict proba and coef
"Logit": {

# For logistic regression, it has built-in predict proba and coef
logit_dict = {
"class_path": "sklearn.linear_model.LogisticRegression",
"default_params": {"C": 1.0, "penalty": "l2", "solver": "liblinear"},
"params_dist": {"C": C, "penalty": ["l2"], "solver": ["liblinear"]}
},
# Use SVC with kernel linear and not LinearSVC, since the latter do not have predict_proba
"Linear_SVM": {
}
# Use SVC with kernel linear and not LinearSVC, since the latter do not have predict_proba
linear_svm_dict ={
"class_path": "sklearn.svm.SVC",
"default_params": {"C": 1.0, "kernel": "linear", "random_state": random_state, "probability": probability},
"params_dist": {"C": C, "kernel": ["linear"] }
},
# Decision Tree have risk of overfitting, hence require pruning of trees
"Decision_Tree": {
}
# Decision Tree classifier, risk of overfitting, hence require pruning of trees
decision_tree_dict = {
"class_path": "sklearn.tree.DecisionTreeClassifier",
"default_params": {"max_depth": 10, "random_state": random_state},
"params_dist": {
Expand All @@ -38,29 +38,20 @@ def load_classifier_class(model_name, random_state=42, probability=True):
"min_samples_leaf": min_sample_leaf,
"max_leaf_nodes": [10, 100, 1000, None]
}
},
# RandomForest shuold in general work better than single decision tree
"Random_Forest": {
}
# Random Forest classifier, should in general work better than single decision tree
random_forest_dict = {
"class_path": "sklearn.ensemble.RandomForestClassifier",
"default_params": {"n_estimators": 10, "max_features": "sqrt", "max_depth": 10, "random_state": random_state, "n_jobs": -1},
"params_dist": {
"max_features": max_features,
"max_leaf_nodes": [10, 100, 1000, None],
"min_samples_leaf": min_sample_leaf
}
},
# AdaBoost is a simpler boosting algorithm
"AdaBoost": {
"class_path": "sklearn.ensemble.AdaBoostClassifier",
"default_params": {"algorithm": "SAMME", "random_state": random_state},
"params_dist": {
"n_estimators": n_estimators,
"learning_rate": learning_rate,
"algorithm": ['SAMME', 'SAMME.R']
}
},
# GradientBoost should tune large number of estimators with slow learning rate
"GradientBoost": {
}

# Gradient Boost classifier, should tune large number of estimators with slow learning rate
gradient_boost_dict = {
"class_path": "sklearn.ensemble.GradientBoostingClassifier",
"default_params": {"loss":'log_loss', "learning_rate":0.1, "n_estimators":10},
"params_dist": {
Expand All @@ -69,7 +60,38 @@ def load_classifier_class(model_name, random_state=42, probability=True):
"min_samples_leaf": min_sample_leaf,
"max_features": max_features
}
}
}
# MLPClassifier is a neural network classifier
mlp_dict = {
"class_path": "sklearn.neural_network.MLPClassifier",
"default_params": {"hidden_layer_sizes": (64,), "max_iter": 500,
"random_state": random_state},
"params_dist": {"hidden_layer_sizes": [(32,), (64,), (128,), (64, 32)],
"alpha": stats.loguniform(1e-5, 1e-1)}
}
# Mimic xgboost
hist_gradient_boost_dict = {
"class_path": "sklearn.ensemble.HistGradientBoostingClassifier",
"default_params": {"random_state": random_state},
"params_dist": {"learning_rate": learning_rate,
"max_leaf_nodes": [15, 31, 63],
"min_samples_leaf": stats.randint(5, 30)}
}

# ========================
# Lastly merge all together into a single dictionary
model_info = {
"Logit": logit_dict,
"Linear_SVM": linear_svm_dict,
# Decision Tree have risk of overfitting, hence require pruning of trees
"Decision_Tree": decision_tree_dict ,
# RandomForest shuold in general work better than single decision tree
"Random_Forest": random_forest_dict,
# GradientBoost should tune large number of estimators with slow learning rate
"Gradient_Boost": gradient_boost_dict,
# MLPClassifier is a neural network classifier
"MLP": mlp_dict,
"Hist_Gradient_Boost": hist_gradient_boost_dict
}

# Check if valid name of model was input
Expand Down
Original file line number Diff line number Diff line change
@@ -1,11 +1,16 @@
from sklearn.model_selection import RandomizedSearchCV
from sklearn.preprocessing import StandardScaler
from sklearn.pipeline import make_pipeline


# Run a randomized search cv to find optimal params
def run_random_search_cv(clf_instance, X, Y, param_distributions, n_iter=10, random_state=42, n_jobs=-1):
# Apply random search cross validation to find optimal hyperparameters
# Make sure to use a pipeline to standardize the data before fitting the model
pipeline = make_pipeline(StandardScaler(), clf_instance)
# Default to use all processors to speed up process
clf_cv = RandomizedSearchCV(
clf_instance, param_distributions=param_distributions,
pipeline, param_distributions=param_distributions,
n_iter=n_iter, random_state=random_state,
n_jobs=n_jobs
)
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -79,7 +79,7 @@ def main(mu_path, dataset_name, model_name, block_num=0, n_iter=10, random_state
print(opt_clf_instance)
# Apply scaling and fit final model
opt_clf = make_pipeline(StandardScaler(), opt_clf_instance)
opt_clf.fit(X_df, y_df["response"])
opt_clf.fit(X_df, y_df["response"].ravel())
# Extract the classifier from the pipeline
classifier = opt_clf.steps[-1][1] # Adjust this based on your pipeline's step name
# Then could either extract their weights or feature importance
Expand Down
2 changes: 1 addition & 1 deletion subworkflows/methods/sklearn/main.nf
Original file line number Diff line number Diff line change
Expand Up @@ -33,7 +33,7 @@ workflow SKLEARN {
// Classifier to train for sklearn
model_name = Channel.fromList(params.sklearn_classifier_names)
// Reduction method (pca or empty)
reduction = Channel.value(params.sklearn_reduction)
reduction = Channel.fromList(params.sklearn_reduction)
take:
// TODO: rename this data_copy to mae_copy or mu_copy depending on language
data_copy // ch of tuple dataset, path of mae/mu data,
Expand Down