Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
24 commits
Select commit Hold shift + click to select a range
138cd5a
Merge pull request #110 from CompBio-Lab/dev
tonyliang19 Feb 28, 2026
85b2b54
Merge pull request #111 from CompBio-Lab/dev
tonyliang19 Mar 10, 2026
e308292
fix: move container option to process level config, and remove stale …
tonyliang19 Aug 13, 2026
b0c6ba5
fix: moved all container definition into config
tonyliang19 Aug 17, 2026
09c4905
fix: changed wrong syntax, adding args to pass in splitting modules, …
tonyliang19 Aug 31, 2026
0819287
fix: lint conf syntax, included back sklearn
tonyliang19 Aug 31, 2026
5f7cd2e
rm: remove old files
tonyliang19 Aug 31, 2026
0ac61ac
rm: remove redundant staled configs
tonyliang19 Aug 31, 2026
2da5e23
fix: rename metric labels
tonyliang19 Aug 31, 2026
939ac4c
rm: remove deprecated workflows
tonyliang19 Aug 31, 2026
7944abd
feat: sklearn to use pca50 or no reduction for more models
tonyliang19 Aug 31, 2026
ec7e648
fix: update method options
tonyliang19 Aug 31, 2026
e5ef195
fix: include select feature for sklearn models
tonyliang19 Sep 3, 2026
7c99113
fix: correct sklearn reduction to list param rather than singleton
tonyliang19 Sep 3, 2026
6f0ee70
fix: correct naming of parameter opt
tonyliang19 Sep 3, 2026
0ce67f6
fix: make sure sklearn cv search works even with make_pipeline
tonyliang19 Sep 4, 2026
b66e821
fix: use eipy container for sklearn as it contains xgboost, and use r…
tonyliang19 Sep 11, 2026
e190ae8
fix: moved common sklearn utils to bin/python_utils, and standardize …
tonyliang19 Sep 11, 2026
e0956f7
fix: refactor code for sklearn select feature
tonyliang19 Sep 11, 2026
44547e9
feat: update sklearn random search cv params as roc
tonyliang19 Sep 15, 2026
36a5963
fix: finalize models in sklearn
tonyliang19 Sep 15, 2026
09fbe25
feat: add new option of sklearn single modality
tonyliang19 Sep 15, 2026
eaefa01
feat: include functionality to handle single modality running, limite…
tonyliang19 Sep 15, 2026
43753b9
feat: include missing scripts of splitting modalities
tonyliang19 Sep 15, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
46 changes: 46 additions & 0 deletions bin/python_utils/combine_mdata2df.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,46 @@
import mudata
import pandas as pd
import numpy as np


# def combine_mdata2df(mdata, concat=True):

# mod_names = list(mdata.mod.keys())

# X_df = pd.concat( [mdata[k].to_df().add_prefix(f"{k}_") for k in mod_names], axis=1 )
# # Also extract the observation df
# y_df = mdata[mod_names[0]].obs
# # Should contain response column and is of string yes or no
# assert y_df["response"].isin(['yes', 'no']).all(), "Column contains values other than 'yes' or 'no'"
# # Converting to numeric binary
# y_df.loc[:, "response"] = np.where(y_df["response"] == "yes", 1, 0)
# # When supplied not to concat meaning y requires other informations more than response
# if not concat:
# return X_df, y_df
# # Merge all DataFrames together, and drop irrevelant meta information
# merged_df = pd.concat([ X_df, y_df[["response"]] ] , axis=1)
# return merged_df, mod_names


def combine_mdata2df(mdata, target_col="response"):
# Takes in a mudata and extract X and y components
# Get the Xs as a dataframe of combining all modality together columnwise
mod_names = list(mdata.mod.keys())
# We also add the modality in front of every feature just like "epigenomics_some_feature_name"
X_df = pd.concat(
[
mdata[mod].to_df().add_prefix(f"{mod}_")
for mod in mod_names
],
axis=1,
)
# Also extract the observation df
y_df = mdata[mod_names[0]].obs.copy()

assert y_df[target_col].isin(["yes", "no"]).all(), (
"Column contains values other than 'yes' or 'no'"
)
# Converting to numeric binary
y_df[target_col] = np.where(y_df[target_col] == "yes", 1, 0)

return X_df, y_df, mod_names
106 changes: 63 additions & 43 deletions ...esources/usr/bin/load_classifier_class.py → bin/python_utils/load_classifier_class.py
100644 → 100755
Original file line number Diff line number Diff line change
@@ -1,6 +1,11 @@
# This script is imported for sklearn classifiers usage


import scipy.stats as stats
import importlib

print("This date should be shown as : 2026")


def load_classifier_class(model_name, random_state=42, probability=True):
# Define models with their respective classes, default parameters, and distributions
Expand All @@ -9,67 +14,82 @@ def load_classifier_class(model_name, random_state=42, probability=True):
# https://stackoverflow.com/questions/33843981/under-what-parameters-are-svc-and-linearsvc-in-scikit-learn-equivalent

# Common params
C = stats.loguniform(1e-4, 1e4)
min_sample_leaf = stats.randint(1,6)
n_estimators = stats.randint(50, 501)
learning_rate = stats.uniform(0.01, 1.1)
max_features = ["sqrt", "log2", 100, 500, 1000, None]
C = stats.loguniform(1e-4, 1e4)
min_sample_leaf = stats.randint(1,6)
base_n_estimators = 10
n_estimators = stats.randint(50, 501)
learning_rate = stats.uniform(0.01, 1.1)
max_features = ["sqrt", "log2", 100, 500, 1000, None]
max_depth = 10



# Dict to store relevant information of sklearn classifiers
model_info = {
# Logistic regression has built-in predict proba and coef
"Logit": {

# For logistic regression, it has built-in predict proba and coef
logit_dict = {
"class_path": "sklearn.linear_model.LogisticRegression",
"default_params": {"C": 1.0, "penalty": "l2", "solver": "liblinear"},
"params_dist": {"C": C, "penalty": ["l2"], "solver": ["liblinear"]}
},
# Use SVC with kernel linear and not LinearSVC, since the latter do not have predict_proba
"Linear_SVM": {
}
# Use SVC with kernel linear and not LinearSVC, since the latter do not have predict_proba
linear_svm_dict ={
"class_path": "sklearn.svm.SVC",
"default_params": {"C": 1.0, "kernel": "linear", "random_state": random_state, "probability": probability},
"params_dist": {"C": C, "kernel": ["linear"] }
},
# Decision Tree have risk of overfitting, hence require pruning of trees
"Decision_Tree": {
"class_path": "sklearn.tree.DecisionTreeClassifier",
"default_params": {"max_depth": 10, "random_state": random_state},
"params_dist": {
"criterion": ['gini', 'entropy', 'log_loss'],
"max_depth": stats.randint(5, 41),
"min_samples_leaf": min_sample_leaf,
"max_leaf_nodes": [10, 100, 1000, None]
}
},
# RandomForest shuold in general work better than single decision tree
"Random_Forest": {
}
# Random Forest classifier, should in general work better than single decision tree
random_forest_dict = {
"class_path": "sklearn.ensemble.RandomForestClassifier",
"default_params": {"n_estimators": 10, "max_features": "sqrt", "max_depth": 10, "random_state": random_state, "n_jobs": -1},
"default_params": {
"n_estimators": base_n_estimators, "max_features": "sqrt",
"max_depth": max_depth,
"random_state": random_state, "n_jobs": -1
},
"params_dist": {
"max_features": max_features,
"max_leaf_nodes": [10, 100, 1000, None],
"min_samples_leaf": min_sample_leaf
}
},
# AdaBoost is a simpler boosting algorithm
"AdaBoost": {
"class_path": "sklearn.ensemble.AdaBoostClassifier",
"default_params": {"algorithm": "SAMME", "random_state": random_state},
}

# MLPClassifier is a neural network classifier
mlp_dict = {
"class_path": "sklearn.neural_network.MLPClassifier",
"default_params": {
"hidden_layer_sizes": (64,), "max_iter": 500,
"random_state": random_state
},
"params_dist": {
"n_estimators": n_estimators,
"learning_rate": learning_rate,
"algorithm": ['SAMME', 'SAMME.R']
"hidden_layer_sizes": [(32,), (64,), (128,), (64, 32)],
"alpha": stats.loguniform(1e-5, 1e-1)
}
},
# GradientBoost should tune large number of estimators with slow learning rate
"GradientBoost": {
"class_path": "sklearn.ensemble.GradientBoostingClassifier",
"default_params": {"loss":'log_loss', "learning_rate":0.1, "n_estimators":10},
}

# Real xgboost
xgboost_dict = {
"class_path": "xgboost.XGBClassifier",
"default_params": {
"random_state": random_state ,
"n_estimators": base_n_estimators,
"objective": "binary:logistic"},
"params_dist": {
"n_estimators": n_estimators,
"learning_rate": learning_rate,
"min_samples_leaf": min_sample_leaf,
"max_features": max_features
"max_depth": stats.randint(2, 9)
}
}
}

# ========================
# Lastly merge all together into a single dictionary
model_info = {
"Logit": logit_dict,
"Linear_SVM": linear_svm_dict,
# RandomForest should in general work better than single decision tree
"Random_Forest": random_forest_dict,
# MLPClassifier is a neural network classifier
"MLP": mlp_dict,
# XGBoost goes here
"XGBoost": xgboost_dict
}

# Check if valid name of model was input
Expand Down
2 changes: 1 addition & 1 deletion conf/real_data.config
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,7 @@ params {
skip_mogonet = false
skip_mofa = false
// Run feature selection
select_feature = true
selectFeature = true
k_fold_number = 5
filter_low_var = "1" // This get casted as bool in both Python and R
// number of component
Expand Down
8 changes: 2 additions & 6 deletions modules/calculate_metrics/main.nf
Original file line number Diff line number Diff line change
@@ -1,11 +1,7 @@
process CALCULATE_METRICS {
def onSockeye = workflow.projectDir.toString().contains('/scratch')
debug true
label 'process_single'
container "${ onSockeye ?
'mogonet.sif' :
'tonyliang19/mogonet:latest' }"

label 'mogonet' // Should be a generic python container?
publishDir (
path: "${params.outdir}/${task.process.tokenize(':').join('/').toLowerCase()}",
mode: 'copy',
Expand All @@ -28,4 +24,4 @@ process CALCULATE_METRICS {
--threshold=${threshold} > \
${task.process.tokenize(':')[-1].toLowerCase()}.log
"""
}
}
32 changes: 19 additions & 13 deletions modules/calculate_metrics/resources/usr/bin/calculate_metrics.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,29 +42,35 @@ def calculate_metrics(group, threshold, average='binary', round_digit=3):
y_pred = np.where(phat >= threshold, 1, 0)
# Initialize dict for metrics
metrics = {
'auc': 0.0,
'roc_auc': 0.0,
'pr_auc': 0.0,
'f1_score': 0.0,
'log_loss': 0.0,
'matthews_corrcoef': 0.0,
'accuracy': 0.0,
'balanced_accuracy': 0.0,
'precision': 0.0,
'average_precision_score': 0.0,
'recall': 0.0,
'f1_score': 0.0,
'log_loss': 0.0,

}

try:
# Check if both classes are present in the current group
if len(set(y_true)) > 1: # More than one unique value in y_true
# AUC requires predicted probability (phat) not y_pred
metrics['auc'] = roc_auc_score(y_true, phat)
metrics['accuracy'] = accuracy_score(y_true, y_pred)
metrics['balanced_accuracy'] = balanced_accuracy_score(y_true, y_pred)
metrics['precision'] = precision_score(y_true, y_pred, average=average)
metrics['average_precision_score'] = average_precision_score(y_true, y_pred)
metrics['recall'] = recall_score(y_true, y_pred, average=average)
metrics['f1_score'] = f1_score(y_true, y_pred, average=average)
# For the losses as well
metrics['log_loss'] = log_loss(y_true, y_pred)
metrics['roc_auc'] = roc_auc_score(y_true, phat)
# Same thing with precision-recall auc
metrics['pr_auc'] = average_precision_score(y_true, phat)
# Rest could use the predicted class (y_pred) to calculate metrics
metrics['f1_score'] = f1_score(y_true, y_pred, average=average)
metrics['log_loss'] = log_loss(y_true, phat)
metrics['matthews_corrcoef'] = matthews_corrcoef(y_true, y_pred)
metrics['accuracy'] = accuracy_score(y_true, y_pred)
metrics['balanced_accuracy'] = balanced_accuracy_score(y_true, y_pred)
metrics['precision'] = precision_score(y_true, y_pred, average=average)
metrics['recall'] = recall_score(y_true, y_pred, average=average)


else:
raise ValueError("Only one class present")

Expand Down
5 changes: 1 addition & 4 deletions modules/caret_multimodal/predict/main.nf
Original file line number Diff line number Diff line change
Expand Up @@ -3,12 +3,9 @@ include { getPublishPath } from "${modulesDir}/functions"

process CARET_MULTIMODAL_PREDICT {
// Vars stuff
def onSockeye = workflow.projectDir.toString().contains('/scratch')
tag "${dataset_name}-${fold_name}"
debug "${params.debug}"
container "${ onSockeye ?
'caret_multimodal.sif' :
'tonyliang19/caret_multimodal:latest' }"
label 'caret_multimodal'

publishDir (
path: "${params.outdir}/${getPublishPath(task.process)}/${dataset_name}/${fold_name}",
Expand Down
7 changes: 1 addition & 6 deletions modules/caret_multimodal/preprocess/main.nf
Original file line number Diff line number Diff line change
Expand Up @@ -21,17 +21,12 @@ include { getPublishPath } from "${modulesDir}/functions"

// TODO: Replace name here
process CARET_MULTIMODAL_PREPROCESS {
// temp variables to use
def onSockeye = workflow.projectDir.toString().contains('/scratch')
// process level configuration
debug "${params.debug}" // debugs true or false by param in MESSI.config
tag "${dataset_name}" // identifier of process when ran in parallel
// By var before to determine what container to use
// Uses apptainer if true otherwise docker
container "${ onSockeye ?
'caret_multimodal.sif' :
'tonyliang19/caret_multimodal:latest' }"

label "caret_multimodal"
label "process_low"

/* outputs store to outdir for saving and inspecting purpose */
Expand Down
5 changes: 1 addition & 4 deletions modules/caret_multimodal/select_feature/main.nf
Original file line number Diff line number Diff line change
Expand Up @@ -4,13 +4,10 @@ include { getPublishPath } from "${modulesDir}/functions"

process CARET_MULTIMODAL_SELECT_FEATURE {
// Vars stuff
def onSockeye = workflow.projectDir.toString().contains('/scratch')
tag "${dataset_name}"
debug true
label 'process_high'
container "${ onSockeye ?
'caret_multimodal.sif' :
'tonyliang19/caret_multimodal:latest' }"
label 'caret_multimodal'

publishDir (
path: "${params.outdir}/${task.process.tokenize(':').join('/').toLowerCase()}/${dataset_name}",
Expand Down
8 changes: 2 additions & 6 deletions modules/caret_multimodal/train/main.nf
Original file line number Diff line number Diff line change
Expand Up @@ -17,14 +17,9 @@
include { getPublishPath } from "${modulesDir}/functions"

process CARET_MULTIMODAL_TRAIN {
// Vars stuff
def onSockeye = workflow.projectDir.toString().contains('/scratch')
// Identifier for each dataset and fold combination
tag "${dataset_name}-${fold_path.name}"
// TODO: rename to the actual image name used
container "${ onSockeye ?
'caret_multimodal.sif' :
'tonyliang19/caret_multimodal:latest' }"

// Parse the output directory to migrate results to
publishDir (
path: "${params.outdir}/${getPublishPath(task.process)}/${dataset_name}/${fold_path.name}",
Expand All @@ -34,6 +29,7 @@ process CARET_MULTIMODAL_TRAIN {
// TODO: add your custom labels to tell if process
// consumes large RAM/ROM and gpu access or not
label 'process_medium'
label 'caret_multimodal'


// TODO: Change arg name to mae_path or mu_path
Expand Down
5 changes: 1 addition & 4 deletions modules/cooperative_learning/predict/main.nf
Original file line number Diff line number Diff line change
Expand Up @@ -2,13 +2,10 @@
include { getPublishPath } from "${modulesDir}/functions"
process COOPERATIVE_LEARNING_PREDICT {
// Vars stuff
def onSockeye = workflow.projectDir.toString().contains('/scratch')
tag "${dataset_name}-${fold_name}"
debug true
label 'process_single'
container "${ onSockeye ?
'codia.sif' :
'tonyliang19/codia:latest' }"
label 'codia'

publishDir (
path: "${params.outdir}/${getPublishPath(task.process)}/${dataset_name}/${fold_name}",
Expand Down
5 changes: 1 addition & 4 deletions modules/cooperative_learning/preprocess/main.nf
Original file line number Diff line number Diff line change
Expand Up @@ -5,14 +5,11 @@ include { getPublishPath } from "${modulesDir}/functions"

process COOPERATIVE_LEARNING_PREPROCESS {
// Temp variables
def onSockeye = workflow.projectDir.toString().contains('/scratch')
/* Directives for process */
debug "${params.debug}" // default is true
tag "${dataset_name}"
label 'process_single'
container "${ onSockeye ?
'codia.sif' :
'tonyliang19/codia:latest'}"
label 'codia'

// More special publish dir by getting method/method_abc to method/abc
publishDir (
Expand Down
7 changes: 2 additions & 5 deletions modules/cooperative_learning/select_feature/main.nf
Original file line number Diff line number Diff line change
Expand Up @@ -4,13 +4,10 @@ include { getPublishPath } from "${modulesDir}/functions"

process COOPERATIVE_LEARNING_SELECT_FEATURE {
// Vars stuff
def onSockeye = workflow.projectDir.toString().contains('/scratch')
tag "${dataset_name}"
debug true
label 'process_high'
container "${ onSockeye ?
'codia.sif' :
'tonyliang19/codia:latest' }"
label 'codia'

publishDir (
path: "${params.outdir}/${task.process.tokenize(':').join('/').toLowerCase()}/${dataset_name}",
Expand All @@ -21,7 +18,7 @@ process COOPERATIVE_LEARNING_SELECT_FEATURE {
// Labels
label 'low_mem'
label 'cpu'
label 'codia'
label 'codia'


/* Input and output blocks*/
Expand Down
Loading
Loading