Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
31 commits
Select commit Hold shift + click to select a range
138cd5a
Merge pull request #110 from CompBio-Lab/dev
tonyliang19 Feb 28, 2026
85b2b54
Merge pull request #111 from CompBio-Lab/dev
tonyliang19 Mar 10, 2026
e308292
fix: move container option to process level config, and remove stale …
tonyliang19 Aug 13, 2026
b0c6ba5
fix: moved all container definition into config
tonyliang19 Aug 17, 2026
09c4905
fix: changed wrong syntax, adding args to pass in splitting modules, …
tonyliang19 Aug 31, 2026
0819287
fix: lint conf syntax, included back sklearn
tonyliang19 Aug 31, 2026
5f7cd2e
rm: remove old files
tonyliang19 Aug 31, 2026
0ac61ac
rm: remove redundant staled configs
tonyliang19 Aug 31, 2026
2da5e23
fix: rename metric labels
tonyliang19 Aug 31, 2026
939ac4c
rm: remove deprecated workflows
tonyliang19 Aug 31, 2026
7944abd
feat: sklearn to use pca50 or no reduction for more models
tonyliang19 Aug 31, 2026
ec7e648
fix: update method options
tonyliang19 Aug 31, 2026
e5ef195
fix: include select feature for sklearn models
tonyliang19 Sep 3, 2026
7c99113
fix: correct sklearn reduction to list param rather than singleton
tonyliang19 Sep 3, 2026
6f0ee70
fix: correct naming of parameter opt
tonyliang19 Sep 3, 2026
0ce67f6
fix: make sure sklearn cv search works even with make_pipeline
tonyliang19 Sep 4, 2026
b66e821
fix: use eipy container for sklearn as it contains xgboost, and use r…
tonyliang19 Sep 11, 2026
e190ae8
fix: moved common sklearn utils to bin/python_utils, and standardize …
tonyliang19 Sep 11, 2026
e0956f7
fix: refactor code for sklearn select feature
tonyliang19 Sep 11, 2026
44547e9
feat: update sklearn random search cv params as roc
tonyliang19 Sep 15, 2026
36a5963
fix: finalize models in sklearn
tonyliang19 Sep 15, 2026
09fbe25
feat: add new option of sklearn single modality
tonyliang19 Sep 15, 2026
eaefa01
feat: include functionality to handle single modality running, limite…
tonyliang19 Sep 15, 2026
43753b9
feat: include missing scripts of splitting modalities
tonyliang19 Sep 15, 2026
9245afc
fix: coerce scheme of rgcca and diablo to 'horst'
tonyliang19 Sep 26, 2026
1fbaaeb
fix: make sure modality names are valid when exporting to filenames
tonyliang19 Sep 26, 2026
06ede3a
feat: add new boolean to run classification or survival
tonyliang19 Sep 26, 2026
8f4a27e
fix: remove old skipper of methods
tonyliang19 Sep 28, 2026
b598a38
feat: add sksurv modules (not tested, need to add container def too
tonyliang19 Sep 28, 2026
6fbc86f
feat: mofa try survival, need to run later
tonyliang19 Sep 28, 2026
bb719ea
feat: this THU specific launcher local script wrapper, del later
tonyliang19 Sep 28, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
46 changes: 46 additions & 0 deletions bin/python_utils/combine_mdata2df.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,46 @@
import mudata
import pandas as pd
import numpy as np


# def combine_mdata2df(mdata, concat=True):

# mod_names = list(mdata.mod.keys())

# X_df = pd.concat( [mdata[k].to_df().add_prefix(f"{k}_") for k in mod_names], axis=1 )
# # Also extract the observation df
# y_df = mdata[mod_names[0]].obs
# # Should contain response column and is of string yes or no
# assert y_df["response"].isin(['yes', 'no']).all(), "Column contains values other than 'yes' or 'no'"
# # Converting to numeric binary
# y_df.loc[:, "response"] = np.where(y_df["response"] == "yes", 1, 0)
# # When supplied not to concat meaning y requires other informations more than response
# if not concat:
# return X_df, y_df
# # Merge all DataFrames together, and drop irrevelant meta information
# merged_df = pd.concat([ X_df, y_df[["response"]] ] , axis=1)
# return merged_df, mod_names


def combine_mdata2df(mdata, target_col="response"):
# Takes in a mudata and extract X and y components
# Get the Xs as a dataframe of combining all modality together columnwise
mod_names = list(mdata.mod.keys())
# We also add the modality in front of every feature just like "epigenomics_some_feature_name"
X_df = pd.concat(
[
mdata[mod].to_df().add_prefix(f"{mod}_")
for mod in mod_names
],
axis=1,
)
# Also extract the observation df
y_df = mdata[mod_names[0]].obs.copy()

assert y_df[target_col].isin(["yes", "no"]).all(), (
"Column contains values other than 'yes' or 'no'"
)
# Converting to numeric binary
y_df[target_col] = np.where(y_df[target_col] == "yes", 1, 0)

return X_df, y_df, mod_names
106 changes: 63 additions & 43 deletions ...esources/usr/bin/load_classifier_class.py → bin/python_utils/load_classifier_class.py
100644 → 100755
Original file line number Diff line number Diff line change
@@ -1,6 +1,11 @@
# This script is imported for sklearn classifiers usage


import scipy.stats as stats
import importlib

print("This date should be shown as : 2026")


def load_classifier_class(model_name, random_state=42, probability=True):
# Define models with their respective classes, default parameters, and distributions
Expand All @@ -9,67 +14,82 @@ def load_classifier_class(model_name, random_state=42, probability=True):
# https://stackoverflow.com/questions/33843981/under-what-parameters-are-svc-and-linearsvc-in-scikit-learn-equivalent

# Common params
C = stats.loguniform(1e-4, 1e4)
min_sample_leaf = stats.randint(1,6)
n_estimators = stats.randint(50, 501)
learning_rate = stats.uniform(0.01, 1.1)
max_features = ["sqrt", "log2", 100, 500, 1000, None]
C = stats.loguniform(1e-4, 1e4)
min_sample_leaf = stats.randint(1,6)
base_n_estimators = 10
n_estimators = stats.randint(50, 501)
learning_rate = stats.uniform(0.01, 1.1)
max_features = ["sqrt", "log2", 100, 500, 1000, None]
max_depth = 10



# Dict to store relevant information of sklearn classifiers
model_info = {
# Logistic regression has built-in predict proba and coef
"Logit": {

# For logistic regression, it has built-in predict proba and coef
logit_dict = {
"class_path": "sklearn.linear_model.LogisticRegression",
"default_params": {"C": 1.0, "penalty": "l2", "solver": "liblinear"},
"params_dist": {"C": C, "penalty": ["l2"], "solver": ["liblinear"]}
},
# Use SVC with kernel linear and not LinearSVC, since the latter do not have predict_proba
"Linear_SVM": {
}
# Use SVC with kernel linear and not LinearSVC, since the latter do not have predict_proba
linear_svm_dict ={
"class_path": "sklearn.svm.SVC",
"default_params": {"C": 1.0, "kernel": "linear", "random_state": random_state, "probability": probability},
"params_dist": {"C": C, "kernel": ["linear"] }
},
# Decision Tree have risk of overfitting, hence require pruning of trees
"Decision_Tree": {
"class_path": "sklearn.tree.DecisionTreeClassifier",
"default_params": {"max_depth": 10, "random_state": random_state},
"params_dist": {
"criterion": ['gini', 'entropy', 'log_loss'],
"max_depth": stats.randint(5, 41),
"min_samples_leaf": min_sample_leaf,
"max_leaf_nodes": [10, 100, 1000, None]
}
},
# RandomForest shuold in general work better than single decision tree
"Random_Forest": {
}
# Random Forest classifier, should in general work better than single decision tree
random_forest_dict = {
"class_path": "sklearn.ensemble.RandomForestClassifier",
"default_params": {"n_estimators": 10, "max_features": "sqrt", "max_depth": 10, "random_state": random_state, "n_jobs": -1},
"default_params": {
"n_estimators": base_n_estimators, "max_features": "sqrt",
"max_depth": max_depth,
"random_state": random_state, "n_jobs": -1
},
"params_dist": {
"max_features": max_features,
"max_leaf_nodes": [10, 100, 1000, None],
"min_samples_leaf": min_sample_leaf
}
},
# AdaBoost is a simpler boosting algorithm
"AdaBoost": {
"class_path": "sklearn.ensemble.AdaBoostClassifier",
"default_params": {"algorithm": "SAMME", "random_state": random_state},
}

# MLPClassifier is a neural network classifier
mlp_dict = {
"class_path": "sklearn.neural_network.MLPClassifier",
"default_params": {
"hidden_layer_sizes": (64,), "max_iter": 500,
"random_state": random_state
},
"params_dist": {
"n_estimators": n_estimators,
"learning_rate": learning_rate,
"algorithm": ['SAMME', 'SAMME.R']
"hidden_layer_sizes": [(32,), (64,), (128,), (64, 32)],
"alpha": stats.loguniform(1e-5, 1e-1)
}
},
# GradientBoost should tune large number of estimators with slow learning rate
"GradientBoost": {
"class_path": "sklearn.ensemble.GradientBoostingClassifier",
"default_params": {"loss":'log_loss', "learning_rate":0.1, "n_estimators":10},
}

# Real xgboost
xgboost_dict = {
"class_path": "xgboost.XGBClassifier",
"default_params": {
"random_state": random_state ,
"n_estimators": base_n_estimators,
"objective": "binary:logistic"},
"params_dist": {
"n_estimators": n_estimators,
"learning_rate": learning_rate,
"min_samples_leaf": min_sample_leaf,
"max_features": max_features
"max_depth": stats.randint(2, 9)
}
}
}

# ========================
# Lastly merge all together into a single dictionary
model_info = {
"Logit": logit_dict,
"Linear_SVM": linear_svm_dict,
# RandomForest should in general work better than single decision tree
"Random_Forest": random_forest_dict,
# MLPClassifier is a neural network classifier
"MLP": mlp_dict,
# XGBoost goes here
"XGBoost": xgboost_dict
}

# Check if valid name of model was input
Expand Down
2 changes: 1 addition & 1 deletion conf/real_data.config
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,7 @@ params {
skip_mogonet = false
skip_mofa = false
// Run feature selection
select_feature = true
selectFeature = true
k_fold_number = 5
filter_low_var = "1" // This get casted as bool in both Python and R
// number of component
Expand Down
76 changes: 76 additions & 0 deletions launcher_local.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,76 @@
#!/bin/bash

# ============================================================================
# This is a wrapper script that triggers a SLURM BATCH script, hence most
# logics are in the other script. Here is more for parsing command line args
#
# Author: Tony Liang
# ============================================================================

# Locates dir of this script
SCRIPT_DIR=$( cd -- "$( dirname -- "${BASH_SOURCE[0]}" )" &> /dev/null && pwd )
LAUNCHER_SCRIPT=${SCRIPT_DIR}/launch_MESSI_pipeline.sh
# The hidden env file
ENV_FILE=${SCRIPT_DIR}/.env
# Check if file exists or not
if [ ! -f ${ENV_FILE} ]; then
echo "missing .env file"
exit 1
else
set -o allexport
source .env # This sources the .env file
set +o allexport
if [ "${ALLOCATION_CODE}" = "REPLACE" ] || [ "${MAIL_USER}" = "REPLACE" ]; then
echo -e "\nERROR: Did not change ALLOCATION_CODE or MAIL_USER\n"
exit 1
fi
fi

# Then call the pipeline here
export NXF_OFFLINE='true'
export NXF_TEMP="/data1/tliang19/tools/nextflow/nf-tmp"
#SELECT_FEAT='true' # Or use 'false'
SELECT_FEAT='false'

#CSV_FILE="data/local_samplesheet.csv"
CSV_FILE="data/samplesheet_bulk.csv"
#CSV_FILE="data/samplesheet_multimodal.csv"
#CSV_FILE="data/samplesheet_htx_only.csv"
#CSV_FILE="data/samplesheet_covid_only.csv"
#CSV_FILE="data/all_datasets.csv"

F="false"
T="true"

FILTER="1" # True in nextflow
#FILTER="0"

#K_FOLD_NUMBER=2
K_FOLD_NUMBER=5
#SK_MODS="Logit,XGBoost"
#SK_MODS="Logit,MLP"

#MAX_MEM="4G"
MAX_MEM="6G"

# Specify number of CPUs and memory
nextflow run main.nf \
-profile standard,docker,test \
--single_modality_mode $T \
--samplesheet ${CSV_FILE} \
--filter_low_var $FILTER \
--outdir results \
--max_memory $MAX_MEM \
--skip_rgcca $F \
--skip_mofa $T \
--skip_sklearn $T \
--skip_caret_multimodal $T \
--skip_diablo $F \
--skip_cplr $T \
--skip_mogonet $T \
--skip_integrao $T \
--pipeline_dir ./ \
--k_fold_number $K_FOLD_NUMBER \
--selectFeature $F \
--publish_relevant $T \
-resume
8 changes: 2 additions & 6 deletions modules/calculate_metrics/main.nf
Original file line number Diff line number Diff line change
@@ -1,11 +1,7 @@
process CALCULATE_METRICS {
def onSockeye = workflow.projectDir.toString().contains('/scratch')
debug true
label 'process_single'
container "${ onSockeye ?
'mogonet.sif' :
'tonyliang19/mogonet:latest' }"

label 'mogonet' // Should be a generic python container?
publishDir (
path: "${params.outdir}/${task.process.tokenize(':').join('/').toLowerCase()}",
mode: 'copy',
Expand All @@ -28,4 +24,4 @@ process CALCULATE_METRICS {
--threshold=${threshold} > \
${task.process.tokenize(':')[-1].toLowerCase()}.log
"""
}
}
32 changes: 19 additions & 13 deletions modules/calculate_metrics/resources/usr/bin/calculate_metrics.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,29 +42,35 @@ def calculate_metrics(group, threshold, average='binary', round_digit=3):
y_pred = np.where(phat >= threshold, 1, 0)
# Initialize dict for metrics
metrics = {
'auc': 0.0,
'roc_auc': 0.0,
'pr_auc': 0.0,
'f1_score': 0.0,
'log_loss': 0.0,
'matthews_corrcoef': 0.0,
'accuracy': 0.0,
'balanced_accuracy': 0.0,
'precision': 0.0,
'average_precision_score': 0.0,
'recall': 0.0,
'f1_score': 0.0,
'log_loss': 0.0,

}

try:
# Check if both classes are present in the current group
if len(set(y_true)) > 1: # More than one unique value in y_true
# AUC requires predicted probability (phat) not y_pred
metrics['auc'] = roc_auc_score(y_true, phat)
metrics['accuracy'] = accuracy_score(y_true, y_pred)
metrics['balanced_accuracy'] = balanced_accuracy_score(y_true, y_pred)
metrics['precision'] = precision_score(y_true, y_pred, average=average)
metrics['average_precision_score'] = average_precision_score(y_true, y_pred)
metrics['recall'] = recall_score(y_true, y_pred, average=average)
metrics['f1_score'] = f1_score(y_true, y_pred, average=average)
# For the losses as well
metrics['log_loss'] = log_loss(y_true, y_pred)
metrics['roc_auc'] = roc_auc_score(y_true, phat)
# Same thing with precision-recall auc
metrics['pr_auc'] = average_precision_score(y_true, phat)
# Rest could use the predicted class (y_pred) to calculate metrics
metrics['f1_score'] = f1_score(y_true, y_pred, average=average)
metrics['log_loss'] = log_loss(y_true, phat)
metrics['matthews_corrcoef'] = matthews_corrcoef(y_true, y_pred)
metrics['accuracy'] = accuracy_score(y_true, y_pred)
metrics['balanced_accuracy'] = balanced_accuracy_score(y_true, y_pred)
metrics['precision'] = precision_score(y_true, y_pred, average=average)
metrics['recall'] = recall_score(y_true, y_pred, average=average)


else:
raise ValueError("Only one class present")

Expand Down
5 changes: 1 addition & 4 deletions modules/caret_multimodal/predict/main.nf
Original file line number Diff line number Diff line change
Expand Up @@ -3,12 +3,9 @@ include { getPublishPath } from "${modulesDir}/functions"

process CARET_MULTIMODAL_PREDICT {
// Vars stuff
def onSockeye = workflow.projectDir.toString().contains('/scratch')
tag "${dataset_name}-${fold_name}"
debug "${params.debug}"
container "${ onSockeye ?
'caret_multimodal.sif' :
'tonyliang19/caret_multimodal:latest' }"
label 'caret_multimodal'

publishDir (
path: "${params.outdir}/${getPublishPath(task.process)}/${dataset_name}/${fold_name}",
Expand Down
7 changes: 1 addition & 6 deletions modules/caret_multimodal/preprocess/main.nf
Original file line number Diff line number Diff line change
Expand Up @@ -21,17 +21,12 @@ include { getPublishPath } from "${modulesDir}/functions"

// TODO: Replace name here
process CARET_MULTIMODAL_PREPROCESS {
// temp variables to use
def onSockeye = workflow.projectDir.toString().contains('/scratch')
// process level configuration
debug "${params.debug}" // debugs true or false by param in MESSI.config
tag "${dataset_name}" // identifier of process when ran in parallel
// By var before to determine what container to use
// Uses apptainer if true otherwise docker
container "${ onSockeye ?
'caret_multimodal.sif' :
'tonyliang19/caret_multimodal:latest' }"

label "caret_multimodal"
label "process_low"

/* outputs store to outdir for saving and inspecting purpose */
Expand Down
5 changes: 1 addition & 4 deletions modules/caret_multimodal/select_feature/main.nf
Original file line number Diff line number Diff line change
Expand Up @@ -4,13 +4,10 @@ include { getPublishPath } from "${modulesDir}/functions"

process CARET_MULTIMODAL_SELECT_FEATURE {
// Vars stuff
def onSockeye = workflow.projectDir.toString().contains('/scratch')
tag "${dataset_name}"
debug true
label 'process_high'
container "${ onSockeye ?
'caret_multimodal.sif' :
'tonyliang19/caret_multimodal:latest' }"
label 'caret_multimodal'

publishDir (
path: "${params.outdir}/${task.process.tokenize(':').join('/').toLowerCase()}/${dataset_name}",
Expand Down
Loading
Loading