Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
19 commits
Select commit Hold shift + click to select a range
1066226
Fix bulk Cirro form params rejected by nf-schema type validation
dltamayo Sep 17, 2026
b6c9f00
Cirro configs: use run_<level> flags instead of deprecated workflow_l…
dltamayo Sep 23, 2026
7b52f8a
Fix run_<level> flags being ignored when passed as false from Cirro
dltamayo Sep 23, 2026
6bbe09f
Fix RENDER_NOTEBOOK reading an unstaged samplesheet path on remote ex…
dltamayo Sep 23, 2026
748dd36
Notebooks: default alias_col to sample name when samplesheet lacks it
dltamayo Sep 23, 2026
6fdddc3
Fix samplesheet_stats.csv losing its row labels
dltamayo Sep 23, 2026
bb20f13
Give RENDER_NOTEBOOK more memory to fix OOM on patient-details report
dltamayo Sep 23, 2026
d453be0
Trim inline comments added in this branch down to one line each
dltamayo Sep 23, 2026
9d98276
Bump RENDER_NOTEBOOK to process_high - process_medium still OOMs
dltamayo Sep 24, 2026
973bc8f
Bump RENDER_NOTEBOOK to process_high_memory (256GB) - still OOMs at 64GB
dltamayo Sep 24, 2026
0385a0c
Vectorize expansion/contraction Fisher's exact tests
dltamayo Sep 24, 2026
65ef10b
Fix RENDER_NOTEBOOK's memory bump never actually applying
dltamayo Sep 24, 2026
3c5aa46
Fix template_discovery_brief's VDJdb section for airr-native input
dltamayo Sep 29, 2026
4ee93c2
Fix public-clones UpSet plot crash on single-patient datasets
dltamayo Sep 29, 2026
a98a69c
Vectorize remaining .apply(axis=1) hotspots in overlap/discovery_brie…
dltamayo Sep 30, 2026
a30f790
RENDER_NOTEBOOK: scale resources per-notebook instead of a flat 16cpu…
dltamayo Sep 30, 2026
89d4f61
RENDER_NOTEBOOK: move resource tiering into conf/modules.config, shri…
dltamayo Sep 30, 2026
0115237
Trim multi-line comments added this session down to one line each
dltamayo Sep 30, 2026
a7e137b
RENDER_NOTEBOOK: move template_details_sample into the 64GB tier
dltamayo Sep 30, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 0 additions & 10 deletions .cirro/bulk_analysis/preprocess.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,14 +25,4 @@
samplesheet.to_csv('samplesheet.csv', index=None)
ds.add_param("samplesheet", "samplesheet.csv")


# 3. Set workflow_level value based on form input
ds.logger.info("Setting workflow_level")

levels = ['convert', 'sample', 'patient', 'compare']
flags = [ds.params['convert_lvl'], ds.params['sample_lvl'], ds.params['patient_lvl'], ds.params['compare_lvl']]
workflow_level = [lvl for lvl, flag in zip(levels, flags) if flag]

ds.add_param('workflow_level', ','.join(workflow_level))

ds.logger.info(ds.params)
7 changes: 3 additions & 4 deletions .cirro/bulk_analysis/process-input.json
Original file line number Diff line number Diff line change
@@ -1,9 +1,8 @@
{
"input_format": "airr",
"convert_lvl": false,
"sample_lvl": "$.params.dataset.paramJson.sample_lvl",
"patient_lvl": "$.params.dataset.paramJson.patient_lvl",
"compare_lvl": "$.params.dataset.paramJson.compare_lvl",
"run_sample": "$.params.dataset.paramJson.sample_lvl",
"run_patient": "$.params.dataset.paramJson.patient_lvl",
"run_compare": "$.params.dataset.paramJson.compare_lvl",
"olga_chunk_length": "$.params.dataset.paramJson.olga_chunk_length",
"matrix_sparsity": "sparse",
"distance_metric": "$.params.dataset.paramJson.distance_metric",
Expand Down
10 changes: 0 additions & 10 deletions .cirro/convert_adaptive/preprocess.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,14 +25,4 @@
samplesheet.to_csv('samplesheet.csv', index=None)
ds.add_param("samplesheet", "samplesheet.csv")


# 3. Set workflow_level value based on form input
ds.logger.info("Setting workflow_level")

levels = ['convert', 'sample', 'compare']
flags = [ds.params['convert_lvl'], ds.params['sample_lvl'], ds.params['compare_lvl']]
workflow_level = [lvl for lvl, flag in zip(levels, flags) if flag]

ds.add_param('workflow_level', ','.join(workflow_level))

ds.logger.info(ds.params)
6 changes: 3 additions & 3 deletions .cirro/convert_adaptive/process-input.json
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
{
"input_format": "adaptive",
"convert_lvl": true,
"sample_lvl": false,
"compare_lvl": false,
"run_convert": true,
"run_sample": false,
"run_compare": false,
"outdir": "$.params.dataset.s3|/data/"
}
2 changes: 1 addition & 1 deletion bin/samplesheet.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@ def samplesheet(samplesheet):
ss.to_csv('samplesheet_utf8.csv', index=False, encoding='utf-8-sig')

stats = ss.describe()
stats.to_csv('samplesheet_stats.csv', index=False, encoding='utf-8-sig')
stats.to_csv('samplesheet_stats.csv', index=True, encoding='utf-8-sig')

print(ss.head())

Expand Down
4 changes: 4 additions & 0 deletions conf/modules.config
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,10 @@ process {
mode: params.publish_dir_mode,
saveAs: { filename -> filename.equals('versions.yml') ? null : filename }
]
// template_details_sample builds dense per-sample TCRdist matrices (OOM'd at 16GB); compare/discovery_brief process full VDJdb match files.
cpus = { ['template_details_sample', 'template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16 * task.attempt : 4 * task.attempt }
memory = { ['template_details_sample', 'template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 64.GB * task.attempt : 16.GB * task.attempt }
time = { ['template_details_sample', 'template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16.h : 4.h }
}

}
5 changes: 3 additions & 2 deletions modules/local/report/render_notebook.nf
Original file line number Diff line number Diff line change
@@ -1,7 +1,6 @@
// Generic process to render a Quarto notebook to HTML
process RENDER_NOTEBOOK {
tag "${notebook.getBaseName()}"
label 'process_single'

input:
// path(files) stages files flat in the root dir; staged_layout optionally
Expand All @@ -16,6 +15,8 @@ process RENDER_NOTEBOOK {
tuple path(notebook), path(files), val(staged_layout)
val project_name
val workflow_cmd
// Fixed name avoids colliding with a same-named file in `files`.
path samplesheet, stageAs: 'render_notebook_samplesheet.csv'

output:
path "${notebook.getBaseName()}.html", emit: report_html
Expand All @@ -34,7 +35,7 @@ process RENDER_NOTEBOOK {
quarto render ${notebook} \\
-P project_name:${project_name} \\
-P workflow_cmd:'${workflow_cmd}' \\
-P sample_table:${file(params.samplesheet)} \\
-P sample_table:${samplesheet} \\
-P subject_col:'${params.subject_col}' \\
-P timepoint_col:'${params.timepoint_col}' \\
-P timepoint_order_col:'${params.timepoint_order_col}' \\
Expand Down
10 changes: 5 additions & 5 deletions nextflow_schema.json
Original file line number Diff line number Diff line change
Expand Up @@ -214,19 +214,19 @@
"default": true
},
"local_min_pvalue": {
"type": "string",
"default": 0.001
"type": ["string", "number"],
"default": "0.001"
},
"simulation_depth": {
"type": "string",
"default": 1000
},
"kmer_min_depth": {
"type": "string",
"default": 3
"type": ["string", "integer"],
"default": "3"
},
"local_min_OVE": {
"type": "string",
"type": ["string", "integer"],
"default": "c(1000, 100, 10)"
},
"all_aa_interchangeable": {
Expand Down
70 changes: 41 additions & 29 deletions notebooks/template_discovery_brief.qmd
Original file line number Diff line number Diff line change
Expand Up @@ -148,6 +148,10 @@ print('Date and time: ' + str(datetime.datetime.now()))

meta = pd.read_csv(sample_table, sep=',')

# alias_col is optional; default to sample name if missing.
if alias_col not in meta.columns:
meta[alias_col] = meta['sample']

# timepoint_order_col isn't a samplesheet column - compute and inject it here.
# timepoint_order (an ordered comma-separated list, e.g. "Base,Week4,EOT") ranks
# timepoints by position in that list; any timepoint not listed sorts after the
Expand Down Expand Up @@ -892,7 +896,7 @@ def create_styled_table(df, title="", max_height='400px', formatter=None):
import pandas as pd
import numpy as np
import itertools
from scipy.stats import fisher_exact
from scipy.stats import hypergeom
from statsmodels.stats.multitest import multipletests

# --- Setup & Pre-computation ---
Expand All @@ -911,11 +915,13 @@ sample_total_counts.rename(columns={'duplicate_count': 'total_counts'}, inplace=
clonotypes_df = pd.merge(clonotypes_df, sample_total_counts, on=[subject_col, 'origin', timepoint_col])
clonotypes_df['frequency'] = clonotypes_df['duplicate_count'] / clonotypes_df['total_counts']

# Helper to run fisher exact test on rows
def run_fisher(row, alt_hyp):
table = [[row['count_post'], row['total_post'] - row['count_post']],
[row['count_pre'], row['total_pre'] - row['count_pre']]]
return fisher_exact(table, alternative=alt_hyp)[1]
# One-sided Fisher's exact test reduces to a single vectorized hypergeometric call since total_pre/total_post are fixed margins.
def run_fisher_vectorized(count_post, count_pre, total_post, total_pre, alt_hyp):
M = total_pre + total_post
n = count_post + count_pre
if alt_hyp == 'greater':
return hypergeom.sf(count_post - 1, M, n, total_post)
return hypergeom.cdf(count_post, M, n, total_post)


# ==========================================
Expand Down Expand Up @@ -968,7 +974,9 @@ for (subject, origin), subject_df in clonotypes_df.groupby([subject_col, 'origin
continue

# 5. Apply Fisher's Exact test only to the remaining rows
comp_df['p_value'] = comp_df.apply(run_fisher, axis=1, alt_hyp='greater')
comp_df['p_value'] = run_fisher_vectorized(
comp_df['count_post'], comp_df['count_pre'], comp_df['total_post'], comp_df['total_pre'], 'greater'
)

# Add metadata
comp_df[subject_col] = subject
Expand Down Expand Up @@ -1090,7 +1098,9 @@ for (subject, origin), subject_df in clonotypes_df.groupby([subject_col, 'origin
continue

# 5. Apply Fisher
comp_df['p_value'] = comp_df.apply(run_fisher, axis=1, alt_hyp='less')
comp_df['p_value'] = run_fisher_vectorized(
comp_df['count_post'], comp_df['count_pre'], comp_df['total_post'], comp_df['total_pre'], 'less'
)

# Add metadata
comp_df[subject_col] = subject
Expand Down Expand Up @@ -1652,17 +1662,12 @@ for subject, subj_df in clonotypes_df.groupby(subject_col):
pivot['freq_t1_plot'] = pivot[t1].replace(0, min_freq)
pivot['freq_t2_plot'] = pivot[t2].replace(0, min_freq)

# Assign highlight status using our sets
def assign_status(row):
key = (subject, origin, t1, t2, row['junction_aa'])
if key in sig_exp_lookup:
return "Expanded"
elif key in sig_cont_lookup:
return "Contracted"
else:
return "Not Significant"

pivot['Status'] = pivot.apply(assign_status, axis=1)
# Avoids the per-row Series-construction overhead of .apply(axis=1).
keys = [(subject, origin, t1, t2, j) for j in pivot['junction_aa']]
pivot['Status'] = [
"Expanded" if key in sig_exp_lookup else ("Contracted" if key in sig_cont_lookup else "Not Significant")
for key in keys
]

# Sort so that colored points render on top of gray points
pivot['sort_order'] = pivot['Status'].map({"Not Significant": 0, "Expanded": 1, "Contracted": 2})
Expand Down Expand Up @@ -1869,24 +1874,27 @@ else:
if len(tps) > 1: return COLOR_MULTI, 'Multiple Timepoints'
else: return timepoint_colors.get(tps[0], '#888888'), tps[0]

node_properties[['color', 'category']] = node_properties['timepoints'].apply(
lambda x: pd.Series(get_color_and_category(x))
)
colors_categories = [get_color_and_category(x) for x in node_properties['timepoints']]
node_properties['color'] = [c for c, _ in colors_categories]
node_properties['category'] = [cat for _, cat in colors_categories]

node_properties['size'] = node_properties['max_frequency'].apply(lambda x: max(10, np.sqrt(x * 100) * 15))
node_properties['size'] = np.maximum(10, np.sqrt(node_properties['max_frequency'] * 100) * 15)

tcr_to_idx = {tcr: i for i, tcr in enumerate(node_properties['cdr3_b_aa'])}

g = ig.Graph()
g.add_vertices(len(node_properties))
g.vs['name'] = node_properties['cdr3_b_aa']
g.vs['size'] = node_properties['size']
g.vs['size'] = node_properties['size']
g.vs['color'] = node_properties['color']
g.vs['category'] = node_properties['category']
g.vs['title'] = node_properties.apply(
lambda row: f"TCR: {row['cdr3_b_aa']}<br>Peak Freq: {(row['max_frequency']*100):.3f}%<br>Peak Count: {row['max_count']}<br>TPs: {row['timepoints']}",
axis=1
).tolist()
g.vs['title'] = [
f"TCR: {cdr3}<br>Peak Freq: {freq*100:.3f}%<br>Peak Count: {count}<br>TPs: {tps}"
for cdr3, freq, count, tps in zip(
node_properties['cdr3_b_aa'], node_properties['max_frequency'],
node_properties['max_count'], node_properties['timepoints']
)
]

patient_samples = patient_df['sample'].unique()
for sample_id in patient_samples:
Expand Down Expand Up @@ -2480,7 +2488,11 @@ def plot_public_clones_upset(concat_df, subject_col, min_shared_patients=2):
# --- 1. Extract Unique Clones per Patient ---
patient_clones = df_clean.groupby(subject_col)['junction_aa'].unique().to_dict()
patient_contents = {str(patient): list(clones) for patient, clones in patient_clones.items()}


if len(patient_contents) < 2:
print("Only one patient present; skipping UpSet plot (needs ≥ 2 patients to show intersections).")
return

# --- 2. Convert to UpSet Multi-Index Format ---
upset_data = from_contents(patient_contents)

Expand Down
5 changes: 5 additions & 0 deletions notebooks/template_giana.qmd
Original file line number Diff line number Diff line change
Expand Up @@ -37,6 +37,11 @@ warnings.filterwarnings(
# Loading data
## reading sample metadata
meta = pd.read_csv(sample_table, sep=',')

# alias_col is optional; default to sample name if missing.
if alias_col not in meta.columns:
meta[alias_col] = meta['sample']

meta_cols = meta.columns.tolist()

```
Expand Down
4 changes: 4 additions & 0 deletions notebooks/template_gliph.qmd
Original file line number Diff line number Diff line change
Expand Up @@ -52,6 +52,10 @@ if timepoint_order_col not in meta.columns:
rank_map = {t: r for r, t in enumerate(sorted(unique_timepoints, key=str))}
meta[timepoint_order_col] = meta[timepoint_col].map(rank_map)

# alias_col is optional; default to sample name if missing.
if alias_col not in meta.columns:
meta[alias_col] = meta['sample']

concat_df = pd.read_csv(concat_csv, sep='\t')
concat_df = concat_df.merge(meta[['sample', 'origin', timepoint_col, timepoint_order_col, alias_col]], on='sample', how='left')

Expand Down
Loading
Loading