Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
68 commits
Select commit Hold shift + click to select a range
0ce67f6
fix: make sure sklearn cv search works even with make_pipeline
tonyliang19 Sep 4, 2026
b66e821
fix: use eipy container for sklearn as it contains xgboost, and use r…
tonyliang19 Sep 11, 2026
e190ae8
fix: moved common sklearn utils to bin/python_utils, and standardize …
tonyliang19 Sep 11, 2026
e0956f7
fix: refactor code for sklearn select feature
tonyliang19 Sep 11, 2026
44547e9
feat: update sklearn random search cv params as roc
tonyliang19 Sep 15, 2026
36a5963
fix: finalize models in sklearn
tonyliang19 Sep 15, 2026
09fbe25
feat: add new option of sklearn single modality
tonyliang19 Sep 15, 2026
eaefa01
feat: include functionality to handle single modality running, limite…
tonyliang19 Sep 15, 2026
43753b9
feat: include missing scripts of splitting modalities
tonyliang19 Sep 15, 2026
9245afc
fix: coerce scheme of rgcca and diablo to 'horst'
tonyliang19 Sep 26, 2026
1fbaaeb
fix: make sure modality names are valid when exporting to filenames
tonyliang19 Sep 26, 2026
06ede3a
feat: add new boolean to run classification or survival
tonyliang19 Sep 26, 2026
8f4a27e
fix: remove old skipper of methods
tonyliang19 Sep 28, 2026
b598a38
feat: add sksurv modules (not tested, need to add container def too
tonyliang19 Sep 28, 2026
6fbc86f
feat: mofa try survival, need to run later
tonyliang19 Sep 28, 2026
bb719ea
feat: this THU specific launcher local script wrapper, del later
tonyliang19 Sep 28, 2026
3c5cca7
Merge pull request #117 from CompBio-Lab/fix-bugs
tonyliang19 Sep 29, 2026
6b06595
Merge pull request #114 from CompBio-Lab/feat-sklearn-pca50
tonyliang19 Sep 29, 2026
86c1816
Merge pull request #115 from CompBio-Lab/feat-sklearn-single-baseline
tonyliang19 Sep 29, 2026
0423fcc
Merge pull request #116 from CompBio-Lab/feat-survival-task-inclusion
tonyliang19 Sep 29, 2026
5ff03de
fix: make sure maxRetries > 1
tonyliang19 Oct 1, 2026
4778266
feat: add missing configs inside real data profile
tonyliang19 Oct 1, 2026
e4a3882
fix: for sockeye to proper work need to enable apptainer within it, a…
tonyliang19 Oct 1, 2026
85338b3
fix: make checking col names / index in mudata more robust
tonyliang19 Oct 1, 2026
6345975
fix: make the local launcher compatible for both sockeye / thu
tonyliang19 Oct 1, 2026
4ad67f6
fix: update the metadata of pipeline, and correctly enabling apptaine…
tonyliang19 Oct 1, 2026
2041601
fix: make the launch pipeline via SBATCH to work in later versions of…
tonyliang19 Oct 1, 2026
3a63020
feat: skip MLP feature importance part and output empty csv for pipel…
tonyliang19 Oct 1, 2026
78c0ce2
fix: if all methods skipped, should output empty csv results for pipe…
tonyliang19 Oct 1, 2026
fac2baa
feat: add common json writers to store method hyperparams
tonyliang19 Oct 2, 2026
01d72ae
feat: add module to merge selected hyperparams from model_selection
tonyliang19 Oct 2, 2026
0e581e5
feat: add caretMultimodal report hyperparms
tonyliang19 Oct 2, 2026
7fd5f75
feat: add cooperativeLearning report hyperparms
tonyliang19 Oct 2, 2026
9413887
feat: add DIABLO report hyperparms
tonyliang19 Oct 2, 2026
26d8efc
feat add IntegrAO report hyperparams
tonyliang19 Oct 2, 2026
e376042
feat add MOGONET report hyperparams
tonyliang19 Oct 2, 2026
2cd5de7
feat add RGCCA report hyperparams
tonyliang19 Oct 2, 2026
1cf25b8
feat add sklearn report hyperparams
tonyliang19 Oct 2, 2026
a3acce8
feat add MOFA report hyperparams
tonyliang19 Oct 2, 2026
d2f3d2f
feat: finalize feature hyperparams report
tonyliang19 Oct 2, 2026
db1c6d4
fix: clean target should also remove subfolders inside work/ after ea…
tonyliang19 Oct 2, 2026
61d6854
fix: make sure to also report selected hyperparameters when want publ…
tonyliang19 Oct 2, 2026
71e75b9
feat: include sksurv as part of pipeline run
tonyliang19 Oct 2, 2026
5f11aa3
feat: add sksurv label into config
tonyliang19 Oct 2, 2026
f5d784f
fix: fallback strategy for sim data on caretMultimodal
tonyliang19 Oct 3, 2026
c77126f
feat: include outcome type to mergers
tonyliang19 Oct 3, 2026
c580fd2
feat: include calc metrics for survival data
tonyliang19 Oct 3, 2026
09e8af4
fix: add fallback of caretMultimodal to not fail with sim data with a…
tonyliang19 Oct 3, 2026
6cfd126
fix: complete the data configs
tonyliang19 Oct 3, 2026
267e028
fix: survival model to work with rsf only
tonyliang19 Oct 3, 2026
15481b8
fix: include outcome type for calculating metrics
tonyliang19 Oct 3, 2026
18e313c
fix: let merge result to parse a file of input of paths instead of li…
tonyliang19 Oct 3, 2026
19780fa
fix: splitting by outcome type
tonyliang19 Oct 3, 2026
31dd400
Merge branch 'dev' of github.com:CompBio-Lab/MESSI-pipeline into dev
tonyliang19 Oct 3, 2026
82179e9
dummy?
tonyliang19 Oct 3, 2026
912f387
merge conf
tonyliang19 Oct 3, 2026
351506d
fix: update the container label for sksurv related modules
tonyliang19 Oct 3, 2026
144bece
Merge branch 'dev' of github.com:CompBio-Lab/MESSI-pipeline into dev
tonyliang19 Oct 4, 2026
f7bc5bb
feat: handle checking survival
tonyliang19 Oct 4, 2026
0aa138e
fix: make sure preprare data not drop columns
tonyliang19 Oct 4, 2026
ef8579d
feat: finalize mofa on survival prediction
tonyliang19 Oct 4, 2026
4982932
feat: skip method dynamically from config opt or if not support other…
tonyliang19 Oct 4, 2026
217c95e
feat: implement cplr to do survival
tonyliang19 Oct 4, 2026
77766d6
fix: update the skipper and include right arg
tonyliang19 Oct 4, 2026
d034011
BUG: cplr not returning right lp?
tonyliang19 Oct 4, 2026
e4b032f
feat: add survival profile
tonyliang19 Oct 4, 2026
3107b0c
feat: include right skippers
tonyliang19 Oct 4, 2026
7139b1e
fix: finalize the launcher scripts
tonyliang19 Oct 9, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -72,6 +72,7 @@ clean:
@rm -rf .nextflow
rm -rf tmp
find work/ -type f -print0 | xargs -0 -P 8 -n 100 rm -f
@rm -rf work
@rm -rf plugins
@rm -rf .config
@rm -f .bash_history
Expand Down
58 changes: 51 additions & 7 deletions bin/misc_utils/checkers.R
Original file line number Diff line number Diff line change
@@ -1,10 +1,54 @@
# Check if matched sample names, then return those
check_common_samples <- function(data) {
sample_names <- lapply(data$X, rownames) |> unlist() |> unique()
matched <- length(sample_names) == length(data$Y)
if (matched) {
return(sample_names)
if (length(data$X) == 0L) {
stop("data$X is empty.")
}

sample_names_list <- lapply(data$X, rownames)

valid_names <- vapply(
sample_names_list,
function(ids) {
!is.null(ids) &&
length(ids) > 0L &&
!anyNA(ids) &&
all(nzchar(ids)) &&
anyDuplicated(ids) == 0L
},
logical(1)
)

if (!all(valid_names)) {
stop("Each modality must have non-empty, unique sample row names.")
}

sample_names <- sample_names_list[[1]]

matched <- vapply(
sample_names_list,
function(ids) identical(ids, sample_names),
logical(1)
)

if (!all(matched)) {
stop("Sample names or their order differ between modalities.")
}

if (is.null(data$Y)) {
stop("data$Y is missing.")
}

y_count <- if (is.null(dim(data$Y))) {
length(data$Y)
} else {
stop("Failed to find match samples, check!")
nrow(data$Y)
}

message("Found ", length(sample_names), " samples in each modality.")
message("Found ", y_count, " observations in Y.")

if (length(sample_names) != y_count) {
stop("Sample count in X does not match observation count in Y.")
}
}

return(sample_names)
}
25 changes: 18 additions & 7 deletions bin/misc_utils/extract_Xy.R
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,7 @@ parseY <- function(Y, verbose = FALSE) {
}

# Convert MAE to list of X and Y
extract_Xy <- function(mae, verbose_target=FALSE) {
extract_Xy <- function(mae, outcome_type="classification", verbose_target=FALSE) {
# Note need to transpose back to p * n
# TODO: add check for dimension match
# COOP LR do not like delayed matrix, so transform it to S3 matrix
Expand All @@ -43,11 +43,22 @@ extract_Xy <- function(mae, verbose_target=FALSE) {
# MAE response would be a dataframe?
# In simulated data this is always atomic, but in real, they are
# transformed to dataframe first, so need to pull it
y_temp <- mae$response
if (is.data.frame(y_temp)) {
Y <- parseY(y_temp |> dplyr::pull(response), verbose = verbose_target)
} else {
Y <- parseY(y_temp, verbose = verbose_target)
if (outcome_type == "survival") {
#stop("Survival outcome type is not implemented yet")
y_df <- SummarizedExperiment::colData(mae) |> as.data.frame()
# And get the time and status columns
out <- y_df[, c("time", "status")]
return(list(X=X, Y=out))
}
if (outcome_type == "classification") {
y_temp <- mae$response
if (is.data.frame(y_temp)) {
Y <- parseY(y_temp |> dplyr::pull(response), verbose = verbose_target)
} else {
Y <- parseY(y_temp, verbose = verbose_target)
}
return(list(X=X, Y=Y))
}
return(list(X=X, Y=Y))
# Otherwise stop
stop("Unsupported outcome_type: ", outcome_type)
}
46 changes: 46 additions & 0 deletions bin/python_utils/combine_mdata2df.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,46 @@
import mudata
import pandas as pd
import numpy as np


# def combine_mdata2df(mdata, concat=True):

# mod_names = list(mdata.mod.keys())

# X_df = pd.concat( [mdata[k].to_df().add_prefix(f"{k}_") for k in mod_names], axis=1 )
# # Also extract the observation df
# y_df = mdata[mod_names[0]].obs
# # Should contain response column and is of string yes or no
# assert y_df["response"].isin(['yes', 'no']).all(), "Column contains values other than 'yes' or 'no'"
# # Converting to numeric binary
# y_df.loc[:, "response"] = np.where(y_df["response"] == "yes", 1, 0)
# # When supplied not to concat meaning y requires other informations more than response
# if not concat:
# return X_df, y_df
# # Merge all DataFrames together, and drop irrevelant meta information
# merged_df = pd.concat([ X_df, y_df[["response"]] ] , axis=1)
# return merged_df, mod_names


def combine_mdata2df(mdata, target_col="response"):
# Takes in a mudata and extract X and y components
# Get the Xs as a dataframe of combining all modality together columnwise
mod_names = list(mdata.mod.keys())
# We also add the modality in front of every feature just like "epigenomics_some_feature_name"
X_df = pd.concat(
[
mdata[mod].to_df().add_prefix(f"{mod}_")
for mod in mod_names
],
axis=1,
)
# Also extract the observation df
y_df = mdata[mod_names[0]].obs.copy()

assert y_df[target_col].isin(["yes", "no"]).all(), (
"Column contains values other than 'yes' or 'no'"
)
# Converting to numeric binary
y_df[target_col] = np.where(y_df[target_col] == "yes", 1, 0)

return X_df, y_df, mod_names
90 changes: 44 additions & 46 deletions ...esources/usr/bin/load_classifier_class.py → bin/python_utils/load_classifier_class.py
100644 → 100755
Original file line number Diff line number Diff line change
@@ -1,6 +1,11 @@
# This script is imported for sklearn classifiers usage


import scipy.stats as stats
import importlib

print("This date should be shown as : 2026")


def load_classifier_class(model_name, random_state=42, probability=True):
# Define models with their respective classes, default parameters, and distributions
Expand All @@ -9,11 +14,16 @@ def load_classifier_class(model_name, random_state=42, probability=True):
# https://stackoverflow.com/questions/33843981/under-what-parameters-are-svc-and-linearsvc-in-scikit-learn-equivalent

# Common params
C = stats.loguniform(1e-4, 1e4)
min_sample_leaf = stats.randint(1,6)
n_estimators = stats.randint(50, 501)
learning_rate = stats.uniform(0.01, 1.1)
max_features = ["sqrt", "log2", 100, 500, 1000, None]
C = stats.loguniform(1e-4, 1e4)
min_sample_leaf = stats.randint(1,6)
base_n_estimators = 10
n_estimators = stats.randint(50, 501)
learning_rate = stats.uniform(0.01, 1.1)
max_features = ["sqrt", "log2", 100, 500, 1000, None]
max_depth = 10



# Dict to store relevant information of sklearn classifiers

# For logistic regression, it has built-in predict proba and coef
Expand All @@ -28,70 +38,58 @@ def load_classifier_class(model_name, random_state=42, probability=True):
"default_params": {"C": 1.0, "kernel": "linear", "random_state": random_state, "probability": probability},
"params_dist": {"C": C, "kernel": ["linear"] }
}
# Decision Tree classifier, risk of overfitting, hence require pruning of trees
decision_tree_dict = {
"class_path": "sklearn.tree.DecisionTreeClassifier",
"default_params": {"max_depth": 10, "random_state": random_state},
"params_dist": {
"criterion": ['gini', 'entropy', 'log_loss'],
"max_depth": stats.randint(5, 41),
"min_samples_leaf": min_sample_leaf,
"max_leaf_nodes": [10, 100, 1000, None]
}
}
# Random Forest classifier, should in general work better than single decision tree
random_forest_dict = {
"class_path": "sklearn.ensemble.RandomForestClassifier",
"default_params": {"n_estimators": 10, "max_features": "sqrt", "max_depth": 10, "random_state": random_state, "n_jobs": -1},
"default_params": {
"n_estimators": base_n_estimators, "max_features": "sqrt",
"max_depth": max_depth,
"random_state": random_state, "n_jobs": -1
},
"params_dist": {
"max_features": max_features,
"max_leaf_nodes": [10, 100, 1000, None],
"min_samples_leaf": min_sample_leaf
}
}

# Gradient Boost classifier, should tune large number of estimators with slow learning rate
gradient_boost_dict = {
"class_path": "sklearn.ensemble.GradientBoostingClassifier",
"default_params": {"loss":'log_loss', "learning_rate":0.1, "n_estimators":10},
"params_dist": {
"n_estimators": n_estimators,
"learning_rate": learning_rate,
"min_samples_leaf": min_sample_leaf,
"max_features": max_features
}
}
# MLPClassifier is a neural network classifier
mlp_dict = {
"class_path": "sklearn.neural_network.MLPClassifier",
"default_params": {"hidden_layer_sizes": (64,), "max_iter": 500,
"random_state": random_state},
"params_dist": {"hidden_layer_sizes": [(32,), (64,), (128,), (64, 32)],
"alpha": stats.loguniform(1e-5, 1e-1)}
"class_path": "sklearn.neural_network.MLPClassifier",
"default_params": {
"hidden_layer_sizes": (64,), "max_iter": 500,
"random_state": random_state
},
"params_dist": {
"hidden_layer_sizes": [(32,), (64,), (128,), (64, 32)],
"alpha": stats.loguniform(1e-5, 1e-1)
}
}
# Mimic xgboost
hist_gradient_boost_dict = {
"class_path": "sklearn.ensemble.HistGradientBoostingClassifier",
"default_params": {"random_state": random_state},
"params_dist": {"learning_rate": learning_rate,
"max_leaf_nodes": [15, 31, 63],
"min_samples_leaf": stats.randint(5, 30)}

# Real xgboost
xgboost_dict = {
"class_path": "xgboost.XGBClassifier",
"default_params": {
"random_state": random_state ,
"n_estimators": base_n_estimators,
"objective": "binary:logistic"},
"params_dist": {
"learning_rate": learning_rate,
"max_depth": stats.randint(2, 9)
}
}

# ========================
# Lastly merge all together into a single dictionary
model_info = {
"Logit": logit_dict,
"Linear_SVM": linear_svm_dict,
# Decision Tree have risk of overfitting, hence require pruning of trees
"Decision_Tree": decision_tree_dict ,
# RandomForest shuold in general work better than single decision tree
# RandomForest should in general work better than single decision tree
"Random_Forest": random_forest_dict,
# GradientBoost should tune large number of estimators with slow learning rate
"Gradient_Boost": gradient_boost_dict,
# MLPClassifier is a neural network classifier
"MLP": mlp_dict,
"Hist_Gradient_Boost": hist_gradient_boost_dict
# XGBoost goes here
"XGBoost": xgboost_dict
}

# Check if valid name of model was input
Expand Down
68 changes: 68 additions & 0 deletions bin/rhelpers.R
Original file line number Diff line number Diff line change
Expand Up @@ -22,3 +22,71 @@ opt2num <- function(opt_chr) {
as.numeric(x), x))
return(opt)
}


make_parameter_record <- function(value, treatment) {
allowed_treatments <- c(
"tuned",
"fixed",
"default",
"data-derived"
)

if (!treatment %in% allowed_treatments) {
stop(
"Unknown parameter treatment: ",
treatment
)
}

list(
value = value,
treatment = treatment
)
}


write_selected_hyperparameters <- function(
dataset_name,
method_name,
parameters,
selection = NULL,
output_path = NULL
) {
if (!requireNamespace("jsonlite", quietly = TRUE)) {
stop("The jsonlite package is required")
}

if (is.null(output_path)) {
safe_method_name <- method_name |>
tolower() |>
gsub("[^a-z0-9_-]+", "_", x = _)

output_path <- paste0(
safe_method_name,
"-",
dataset_name,
"_selected_hyperparameters.json"
)
}

result <- list(
dataset = dataset_name,
method = method_name,
analysis_stage = "model_selection",
selection = selection,
parameters = parameters
)

jsonlite::write_json(
result,
path = output_path,
pretty = TRUE,
auto_unbox = TRUE,
null = "null",
na = "null",
digits = NA
)

invisible(output_path)
}
Loading
Loading