From 106622696f72fffad7d158988932e7ce816cd243 Mon Sep 17 00:00:00 2001 From: dltamayo Date: Thu, 17 Sep 2026 15:19:42 -0400 Subject: [PATCH 01/19] Fix bulk Cirro form params rejected by nf-schema type validation Cirro submits kmer_min_depth, local_min_OVE, and local_min_pvalue as native numeric types when overridden through the bulk analysis form, but nextflow_schema.json declared them strictly as type "string". validateParameters() rejected the run before any process executed (reproduced with `nextflow run . --kmer_min_depth 3 --local_min_OVE 10 --local_min_pvalue 0.001`). Widen the accepted types and fix two defaults that had drifted to unquoted JSON numbers. Co-Authored-By: Claude Sonnet 5 --- nextflow_schema.json | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/nextflow_schema.json b/nextflow_schema.json index 5b7bdbd..8350b27 100644 --- a/nextflow_schema.json +++ b/nextflow_schema.json @@ -214,19 +214,19 @@ "default": true }, "local_min_pvalue": { - "type": "string", - "default": 0.001 + "type": ["string", "number"], + "default": "0.001" }, "simulation_depth": { "type": "string", "default": 1000 }, "kmer_min_depth": { - "type": "string", - "default": 3 + "type": ["string", "integer"], + "default": "3" }, "local_min_OVE": { - "type": "string", + "type": ["string", "integer"], "default": "c(1000, 100, 10)" }, "all_aa_interchangeable": { From b6c9f00917faa35fb5256103c686a9f41f37de5c Mon Sep 17 00:00:00 2001 From: dltamayo Date: Wed, 23 Sep 2026 11:33:54 -0400 Subject: [PATCH 02/19] Cirro configs: use run_ flags instead of deprecated workflow_level The pipeline now warns "workflow_level is set for deprecation, please use run_ flags" on every Cirro run. Both bulk_analysis and convert_adaptive preprocess.py were still building the old comma-joined workflow_level string from convert_lvl/sample_lvl/patient_lvl/compare_lvl. Point process-input.json directly at run_sample/run_patient/run_compare/ run_convert (the params workflows/tcrtoolkit_bulk.nf actually reads) and drop the now-unnecessary workflow_level construction from both scripts. Co-Authored-By: Claude Sonnet 5 --- .cirro/bulk_analysis/preprocess.py | 10 ---------- .cirro/bulk_analysis/process-input.json | 7 +++---- .cirro/convert_adaptive/preprocess.py | 10 ---------- .cirro/convert_adaptive/process-input.json | 6 +++--- 4 files changed, 6 insertions(+), 27 deletions(-) diff --git a/.cirro/bulk_analysis/preprocess.py b/.cirro/bulk_analysis/preprocess.py index 45a4c95..8431046 100644 --- a/.cirro/bulk_analysis/preprocess.py +++ b/.cirro/bulk_analysis/preprocess.py @@ -25,14 +25,4 @@ samplesheet.to_csv('samplesheet.csv', index=None) ds.add_param("samplesheet", "samplesheet.csv") - -# 3. Set workflow_level value based on form input -ds.logger.info("Setting workflow_level") - -levels = ['convert', 'sample', 'patient', 'compare'] -flags = [ds.params['convert_lvl'], ds.params['sample_lvl'], ds.params['patient_lvl'], ds.params['compare_lvl']] -workflow_level = [lvl for lvl, flag in zip(levels, flags) if flag] - -ds.add_param('workflow_level', ','.join(workflow_level)) - ds.logger.info(ds.params) diff --git a/.cirro/bulk_analysis/process-input.json b/.cirro/bulk_analysis/process-input.json index 8ca08df..4cc09ca 100644 --- a/.cirro/bulk_analysis/process-input.json +++ b/.cirro/bulk_analysis/process-input.json @@ -1,9 +1,8 @@ { "input_format": "airr", - "convert_lvl": false, - "sample_lvl": "$.params.dataset.paramJson.sample_lvl", - "patient_lvl": "$.params.dataset.paramJson.patient_lvl", - "compare_lvl": "$.params.dataset.paramJson.compare_lvl", + "run_sample": "$.params.dataset.paramJson.sample_lvl", + "run_patient": "$.params.dataset.paramJson.patient_lvl", + "run_compare": "$.params.dataset.paramJson.compare_lvl", "olga_chunk_length": "$.params.dataset.paramJson.olga_chunk_length", "matrix_sparsity": "sparse", "distance_metric": "$.params.dataset.paramJson.distance_metric", diff --git a/.cirro/convert_adaptive/preprocess.py b/.cirro/convert_adaptive/preprocess.py index ede348a..8431046 100644 --- a/.cirro/convert_adaptive/preprocess.py +++ b/.cirro/convert_adaptive/preprocess.py @@ -25,14 +25,4 @@ samplesheet.to_csv('samplesheet.csv', index=None) ds.add_param("samplesheet", "samplesheet.csv") - -# 3. Set workflow_level value based on form input -ds.logger.info("Setting workflow_level") - -levels = ['convert', 'sample', 'compare'] -flags = [ds.params['convert_lvl'], ds.params['sample_lvl'], ds.params['compare_lvl']] -workflow_level = [lvl for lvl, flag in zip(levels, flags) if flag] - -ds.add_param('workflow_level', ','.join(workflow_level)) - ds.logger.info(ds.params) diff --git a/.cirro/convert_adaptive/process-input.json b/.cirro/convert_adaptive/process-input.json index a5d764f..10f7cc1 100644 --- a/.cirro/convert_adaptive/process-input.json +++ b/.cirro/convert_adaptive/process-input.json @@ -1,7 +1,7 @@ { "input_format": "adaptive", - "convert_lvl": true, - "sample_lvl": false, - "compare_lvl": false, + "run_convert": true, + "run_sample": false, + "run_compare": false, "outdir": "$.params.dataset.s3|/data/" } \ No newline at end of file From 7b52f8aa97f8ef546280e387425a143ba0067889 Mon Sep 17 00:00:00 2001 From: dltamayo Date: Wed, 23 Sep 2026 12:09:11 -0400 Subject: [PATCH 03/19] Fix run_ flags being ignored when passed as false from Cirro Nextflow's CLI param parser auto-casts numeric-looking overrides to Integer/Double but leaves "true"/"false" as plain Strings. Groovy truthiness treats any non-empty String as true, so `if (params.run_sample)` etc. always ran that stage regardless of an explicit false override - verified locally that --run_sample false --run_compare false still launched SAMPLE (VDJDB_GET and friends). This silently broke the convert-only pathway that .cirro/convert_adaptive depends on (run_sample: false, run_compare: false), since Cirro renders these as CLI flags the same way it does numeric params. Route each flag through toString().toBoolean() so both real Booleans and Cirro's string overrides parse correctly. Verified with a real (non-stub, Docker) convert-only run: only SAMPLESHEET_CHECK and CONVERT_ADAPTIVE execute now. Co-Authored-By: Claude Sonnet 5 --- workflows/tcrtoolkit_bulk.nf | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/workflows/tcrtoolkit_bulk.nf b/workflows/tcrtoolkit_bulk.nf index 8a56b20..f06ccbb 100644 --- a/workflows/tcrtoolkit_bulk.nf +++ b/workflows/tcrtoolkit_bulk.nf @@ -23,12 +23,17 @@ workflow TCRTOOLKIT_BULK { println("Running TCRTOOLKIT_BULK workflow...") - // Construct levels list from the run_sample, run_compare, and run_patient parameters + // Construct levels list from the run_sample, run_compare, and run_patient parameters. + // Cirro/Nextflow CLI overrides land here as the string "false", not Boolean false - + // Groovy truthiness treats any non-empty String (including "false") as true, so a + // plain `if (params.run_sample)` would never actually disable a level. Route every + // flag through GDK's String.toBoolean() (via toString(), which also normalizes real + // Booleans and null) to parse the intended value correctly regardless of source. def levels = [] - if (params.run_sample) levels << 'sample' - if (params.run_compare) levels << 'compare' - if (params.run_patient) levels << 'patient' - if (params.run_convert) levels << 'convert' + if (params.run_sample?.toString()?.toBoolean() ?: false) levels << 'sample' + if (params.run_compare?.toString()?.toBoolean() ?: false) levels << 'compare' + if (params.run_patient?.toString()?.toBoolean() ?: false) levels << 'patient' + if (params.run_convert?.toString()?.toBoolean() ?: false) levels << 'convert' def input_format = params.input_format.toLowerCase() From 6bbe09fad74f813624e437216c5f6229c0b21ea2 Mon Sep 17 00:00:00 2001 From: dltamayo Date: Wed, 23 Sep 2026 13:39:17 -0400 Subject: [PATCH 04/19] Fix RENDER_NOTEBOOK reading an unstaged samplesheet path on remote executors RENDER_NOTEBOOK built its `quarto render -P sample_table:...` arg from `${file(params.samplesheet)}` as raw script-string interpolation rather than a declared process input, so Nextflow never staged the samplesheet into the task's work dir - it just embedded the head node's local absolute path. That happened to still resolve when running fully local (head and worker share a filesystem), but on Cirro/AWS Batch the worker container has no such path: dataset 261a4d9a failed with `FileNotFoundError: /opt/work/.../samplesheet.csv` once RENDER_NOTEBOOK (template_qc) ran, after every other stage (sample/patient incl. GIANA/GLIPH2) had completed successfully on fix/bulk-cirro-config. Declare `path samplesheet` as a real process input (stageAs'd to a fixed name so it can't collide with a same-named file in the caller's report_files list, as the "nested project_dir layout" nf-test caught) and reference the staged file directly. Verified via nf-test - the rendered .command.sh now references the staged bare filename, not an absolute host path. Co-Authored-By: Claude Sonnet 5 --- modules/local/report/render_notebook.nf | 7 ++++++- subworkflows/local/report.nf | 3 ++- tests/modules/local/report/render_notebook.nf.test | 3 +++ 3 files changed, 11 insertions(+), 2 deletions(-) diff --git a/modules/local/report/render_notebook.nf b/modules/local/report/render_notebook.nf index b9b6e3b..0c010c1 100644 --- a/modules/local/report/render_notebook.nf +++ b/modules/local/report/render_notebook.nf @@ -16,6 +16,11 @@ process RENDER_NOTEBOOK { tuple path(notebook), path(files), val(staged_layout) val project_name val workflow_cmd + // Staged under a fixed name, distinct from anything in `files`, since a + // caller's report_files list could otherwise legitimately contain a file + // with the same basename as the samplesheet, which Nextflow would refuse + // to stage (input file name collision). + path samplesheet, stageAs: 'render_notebook_samplesheet.csv' output: path "${notebook.getBaseName()}.html", emit: report_html @@ -34,7 +39,7 @@ process RENDER_NOTEBOOK { quarto render ${notebook} \\ -P project_name:${project_name} \\ -P workflow_cmd:'${workflow_cmd}' \\ - -P sample_table:${file(params.samplesheet)} \\ + -P sample_table:${samplesheet} \\ -P subject_col:'${params.subject_col}' \\ -P timepoint_col:'${params.timepoint_col}' \\ -P timepoint_order_col:'${params.timepoint_order_col}' \\ diff --git a/subworkflows/local/report.nf b/subworkflows/local/report.nf index f66c95a..855e1bb 100644 --- a/subworkflows/local/report.nf +++ b/subworkflows/local/report.nf @@ -23,7 +23,8 @@ workflow REPORT { RENDER_NOTEBOOK( ch_reports, params.project_name, - workflow.commandLine + workflow.commandLine, + file(params.samplesheet) ) emit: diff --git a/tests/modules/local/report/render_notebook.nf.test b/tests/modules/local/report/render_notebook.nf.test index 8f7290a..e6cdbb4 100644 --- a/tests/modules/local/report/render_notebook.nf.test +++ b/tests/modules/local/report/render_notebook.nf.test @@ -26,6 +26,7 @@ nextflow_process { ] input[1] = "TCRtoolkit" input[2] = "nextflow run main.nf" + input[3] = file(params.samplesheet) """ } } @@ -63,6 +64,7 @@ nextflow_process { ] input[1] = "TCRtoolkit" input[2] = "nextflow run main.nf" + input[3] = file(params.samplesheet) """ } } @@ -107,6 +109,7 @@ nextflow_process { ] input[1] = "TCRtoolkit" input[2] = "nextflow run main.nf" + input[3] = file(params.samplesheet) """ } } From 748dd364a9a39d996307304f962df9bb6bcab249 Mon Sep 17 00:00:00 2001 From: dltamayo Date: Wed, 23 Sep 2026 14:48:22 -0400 Subject: [PATCH 05/19] Notebooks: default alias_col to sample name when samplesheet lacks it Dataset c2fa7c64 failed at RENDER_NOTEBOOK (template_qc and template_details_patient) with: ValueError: Value of 'hover_data_0' is not the name of a column in 'data_frame'. Expected one of [...] but received: alias alias_col (default "alias") is documented as an optional samplesheet column, and several call sites already guard with `if 'alias' in ...columns`, but each notebook's own `meta = pd.read_csv(sample_table)` never actually backfills it - other spots (hover_data=['alias'], column-selection merges) reference it unconditionally and blow up the instant a real samplesheet omits it, as this one did. Add the same defaulting each notebook already uses for timepoint_order_col ("if col not in meta.columns: inject it") for alias_col too, right after each notebook's own meta load: fall back to the sample name. Applied everywhere a notebook independently loads sample_table and references alias_col (template_qc, template_giana, template_gliph, template_discovery_brief, template_overlap, template_pheno_sc/bulk, template_sample) - not just the two that happened to fail in this run, since it's the same structural gap in every one of them. Verified: the fallback logic re-run against the real container's pandas, and every edited notebook's Python code blocks still compile. Have not re-run the full multi-hour pipeline end-to-end locally to confirm no further downstream issue - recommend retrying on Cirro next. Co-Authored-By: Claude Sonnet 5 --- notebooks/template_discovery_brief.qmd | 5 +++++ notebooks/template_giana.qmd | 6 ++++++ notebooks/template_gliph.qmd | 5 +++++ notebooks/template_overlap.qmd | 5 +++++ notebooks/template_pheno_bulk.qmd | 5 +++++ notebooks/template_pheno_sc.qmd | 5 +++++ notebooks/template_qc.qmd | 5 +++++ notebooks/template_sample.qmd | 5 +++++ 8 files changed, 41 insertions(+) diff --git a/notebooks/template_discovery_brief.qmd b/notebooks/template_discovery_brief.qmd index d0da26d..62d7d81 100644 --- a/notebooks/template_discovery_brief.qmd +++ b/notebooks/template_discovery_brief.qmd @@ -148,6 +148,11 @@ print('Date and time: ' + str(datetime.datetime.now())) meta = pd.read_csv(sample_table, sep=',') +# alias_col is optional in the samplesheet; default to the sample name when +# absent so the alias-keyed columns/labels below always have a value. +if alias_col not in meta.columns: + meta[alias_col] = meta['sample'] + # timepoint_order_col isn't a samplesheet column - compute and inject it here. # timepoint_order (an ordered comma-separated list, e.g. "Base,Week4,EOT") ranks # timepoints by position in that list; any timepoint not listed sorts after the diff --git a/notebooks/template_giana.qmd b/notebooks/template_giana.qmd index 6e2c56c..028b63c 100644 --- a/notebooks/template_giana.qmd +++ b/notebooks/template_giana.qmd @@ -37,6 +37,12 @@ warnings.filterwarnings( # Loading data ## reading sample metadata meta = pd.read_csv(sample_table, sep=',') + +# alias_col is optional in the samplesheet; default to the sample name when +# absent so the alias-keyed groupings/labels below always have a value. +if alias_col not in meta.columns: + meta[alias_col] = meta['sample'] + meta_cols = meta.columns.tolist() ``` diff --git a/notebooks/template_gliph.qmd b/notebooks/template_gliph.qmd index 004975b..cbba151 100644 --- a/notebooks/template_gliph.qmd +++ b/notebooks/template_gliph.qmd @@ -52,6 +52,11 @@ if timepoint_order_col not in meta.columns: rank_map = {t: r for r, t in enumerate(sorted(unique_timepoints, key=str))} meta[timepoint_order_col] = meta[timepoint_col].map(rank_map) +# alias_col is optional in the samplesheet; default to the sample name when +# absent so the alias-keyed columns/labels below always have a value. +if alias_col not in meta.columns: + meta[alias_col] = meta['sample'] + concat_df = pd.read_csv(concat_csv, sep='\t') concat_df = concat_df.merge(meta[['sample', 'origin', timepoint_col, timepoint_order_col, alias_col]], on='sample', how='left') diff --git a/notebooks/template_overlap.qmd b/notebooks/template_overlap.qmd index 7312aac..8b882b3 100644 --- a/notebooks/template_overlap.qmd +++ b/notebooks/template_overlap.qmd @@ -53,6 +53,11 @@ import re ## Reading sample metadata meta = pd.read_csv(sample_table, sep=',') +# alias_col is optional in the samplesheet; default to the sample name when +# absent so the alias-keyed columns/labels below always have a value. +if alias_col not in meta.columns: + meta[alias_col] = meta['sample'] + # timepoint_order_col isn't a samplesheet column - compute and inject it here. # timepoint_order (an ordered comma-separated list, e.g. "Base,Week4,EOT") ranks # timepoints by position in that list; any timepoint not listed sorts after the diff --git a/notebooks/template_pheno_bulk.qmd b/notebooks/template_pheno_bulk.qmd index 624b2cd..48a9ff1 100644 --- a/notebooks/template_pheno_bulk.qmd +++ b/notebooks/template_pheno_bulk.qmd @@ -61,6 +61,11 @@ warnings.filterwarnings( meta = pd.read_csv(sample_table, sep=',') meta.drop(columns=['file'], inplace=True) +# alias_col is optional in the samplesheet; default to the sample name when +# absent so the alias-keyed columns/labels below always have a value. +if alias_col not in meta.columns: + meta[alias_col] = meta['sample'] + # timepoint_order_col isn't a samplesheet column - compute and inject it here. # timepoint_order (an ordered comma-separated list, e.g. "Base,Week4,EOT") ranks # timepoints by position in that list; any timepoint not listed sorts after the diff --git a/notebooks/template_pheno_sc.qmd b/notebooks/template_pheno_sc.qmd index 5d30dc1..5b5e687 100644 --- a/notebooks/template_pheno_sc.qmd +++ b/notebooks/template_pheno_sc.qmd @@ -50,6 +50,11 @@ import re meta = pd.read_csv(sample_table, sep=',') meta.drop(columns=['file'], inplace=True) +# alias_col is optional in the samplesheet; default to the sample name when +# absent so the alias-keyed columns/labels below always have a value. +if alias_col not in meta.columns: + meta[alias_col] = meta['sample'] + # timepoint_order_col isn't a samplesheet column - compute and inject it here. # timepoint_order (an ordered comma-separated list, e.g. "Base,Week4,EOT") ranks # timepoints by position in that list; any timepoint not listed sorts after the diff --git a/notebooks/template_qc.qmd b/notebooks/template_qc.qmd index 471f256..866183e 100644 --- a/notebooks/template_qc.qmd +++ b/notebooks/template_qc.qmd @@ -137,6 +137,11 @@ if timepoint_order_col not in meta.columns: rank_map = {t: r for r, t in enumerate(sorted(unique_timepoints, key=str))} meta[timepoint_order_col] = meta[timepoint_col].map(rank_map) +# alias_col is optional in the samplesheet; default to the sample name when +# absent so hover labels and per-sample groupings below always have a value. +if alias_col not in meta.columns: + meta[alias_col] = meta['sample'] + meta_cols = meta.columns.tolist() df = pd.read_csv(sample_stats_csv, sep=',') diff --git a/notebooks/template_sample.qmd b/notebooks/template_sample.qmd index aea664e..edc420b 100644 --- a/notebooks/template_sample.qmd +++ b/notebooks/template_sample.qmd @@ -50,6 +50,11 @@ warnings.filterwarnings( meta = pd.read_csv(sample_table, sep=',') meta.drop(columns=['file'], inplace=True) +# alias_col is optional in the samplesheet; default to the sample name when +# absent so the alias-keyed columns/labels below always have a value. +if alias_col not in meta.columns: + meta[alias_col] = meta['sample'] + # timepoint_order_col isn't a samplesheet column - compute and inject it here. # timepoint_order (an ordered comma-separated list, e.g. "Base,Week4,EOT") ranks # timepoints by position in that list; any timepoint not listed sorts after the From 6fdddc3aefdd7cf02f83feb3e9c98de6f4684411 Mon Sep 17 00:00:00 2001 From: dltamayo Date: Wed, 23 Sep 2026 14:52:55 -0400 Subject: [PATCH 06/19] Fix samplesheet_stats.csv losing its row labels SAMPLESHEET_CHECK's samplesheet_stats.csv (e.g. dataset 2059e0b4) was unreadable: bin/samplesheet.py builds it from ss.describe() (whose row index holds the stat names - count/unique/top/freq for object columns, or count/mean/std/min/25%/50%/75%/max for numeric ones) but wrote it with index=False, dropping those labels entirely. The CSV ends up with the right values in the right columns but no way to tell which row is which stat. Write it with index=True instead. Verified against the real container: the fixed output now has the count/unique/top/freq labels in the first column, matching what .describe() actually produced. Co-Authored-By: Claude Sonnet 5 --- bin/samplesheet.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/bin/samplesheet.py b/bin/samplesheet.py index eb1e06f..add7260 100755 --- a/bin/samplesheet.py +++ b/bin/samplesheet.py @@ -9,7 +9,7 @@ def samplesheet(samplesheet): ss.to_csv('samplesheet_utf8.csv', index=False, encoding='utf-8-sig') stats = ss.describe() - stats.to_csv('samplesheet_stats.csv', index=False, encoding='utf-8-sig') + stats.to_csv('samplesheet_stats.csv', index=True, encoding='utf-8-sig') print(ss.head()) From bb20f131ccfca0b7baa4a3878712cd9b79e38e4b Mon Sep 17 00:00:00 2001 From: dltamayo Date: Wed, 23 Sep 2026 17:15:23 -0400 Subject: [PATCH 07/19] Give RENDER_NOTEBOOK more memory to fix OOM on patient-details report Dataset a0231533 got through every stage (sample, patient, GIANA, GLIPH2, template_qc render) and OOM-killed on RENDER_NOTEBOOK (template_details_patient): Essential container in task exited - OutOfMemoryError: Container killed due to memory usage RENDER_NOTEBOOK is labeled process_single, which only budgets 2.GB. That's enough for template_qc (which just succeeded in the same run) but not template_details_patient, which embeds GIANA sunburst and GLIPH2 network plots into one HTML file via quarto's embed-resources: true. The pipeline's usual attempt-scaled-memory retry (errorStrategy checks task.exitStatus in (130..145)+[104], which covers a SIGKILL/137 OOM) never engaged either - the task list shows only one attempt with exitCode 1, not 137, so Batch didn't surface the OOM as a retryable code here. Bump it to process_medium (16.GB) via a withName override in conf/modules.config, matching the existing TCRDIST3_MATRIX pattern in the same file. Verified with `nextflow config -flat` that the label override resolves as intended. Co-Authored-By: Claude Sonnet 5 --- conf/modules.config | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/conf/modules.config b/conf/modules.config index 1b959dd..605b1d1 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -24,6 +24,13 @@ process { } withName: RENDER_NOTEBOOK { + // process_single's 2.GB default is enough for lighter reports (e.g. + // template_qc) but not the patient-details report, which embeds GIANA + // sunburst + GLIPH2 network plots via embed-resources: true - dataset + // a0231533 OOM-killed at 2GB. AWS Batch didn't surface that OOM as one + // of the retryable exit codes (task.exitStatus stayed 1, not 137), so + // the usual attempt-scaled-memory retry never kicked in either. + label = 'process_medium' publishDir = [ path: { "${params.outdir}/report" }, mode: params.publish_dir_mode, From d453be0369f524fd8419a0bad8d29aa696e640e2 Mon Sep 17 00:00:00 2001 From: dltamayo Date: Wed, 23 Sep 2026 17:17:17 -0400 Subject: [PATCH 08/19] Trim inline comments added in this branch down to one line each The full rationale for each of these already lives in its own commit message (samplesheet.py stats fix, run_ boolean coercion, RENDER_NOTEBOOK samplesheet staging/collision, alias_col fallback, RENDER_NOTEBOOK memory bump) - no need to duplicate it in the code too. Co-Authored-By: Claude Sonnet 5 --- conf/modules.config | 7 +------ modules/local/report/render_notebook.nf | 5 +---- notebooks/template_discovery_brief.qmd | 3 +-- notebooks/template_giana.qmd | 3 +-- notebooks/template_gliph.qmd | 3 +-- notebooks/template_overlap.qmd | 3 +-- notebooks/template_pheno_bulk.qmd | 3 +-- notebooks/template_pheno_sc.qmd | 3 +-- notebooks/template_qc.qmd | 3 +-- notebooks/template_sample.qmd | 3 +-- workflows/tcrtoolkit_bulk.nf | 7 +------ 11 files changed, 11 insertions(+), 32 deletions(-) diff --git a/conf/modules.config b/conf/modules.config index 605b1d1..3f5f39d 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -24,12 +24,7 @@ process { } withName: RENDER_NOTEBOOK { - // process_single's 2.GB default is enough for lighter reports (e.g. - // template_qc) but not the patient-details report, which embeds GIANA - // sunburst + GLIPH2 network plots via embed-resources: true - dataset - // a0231533 OOM-killed at 2GB. AWS Batch didn't surface that OOM as one - // of the retryable exit codes (task.exitStatus stayed 1, not 137), so - // the usual attempt-scaled-memory retry never kicked in either. + // patient-details report OOMs at process_single's 2.GB default. label = 'process_medium' publishDir = [ path: { "${params.outdir}/report" }, diff --git a/modules/local/report/render_notebook.nf b/modules/local/report/render_notebook.nf index 0c010c1..a25ac23 100644 --- a/modules/local/report/render_notebook.nf +++ b/modules/local/report/render_notebook.nf @@ -16,10 +16,7 @@ process RENDER_NOTEBOOK { tuple path(notebook), path(files), val(staged_layout) val project_name val workflow_cmd - // Staged under a fixed name, distinct from anything in `files`, since a - // caller's report_files list could otherwise legitimately contain a file - // with the same basename as the samplesheet, which Nextflow would refuse - // to stage (input file name collision). + // Fixed name avoids colliding with a same-named file in `files`. path samplesheet, stageAs: 'render_notebook_samplesheet.csv' output: diff --git a/notebooks/template_discovery_brief.qmd b/notebooks/template_discovery_brief.qmd index 62d7d81..8bb7895 100644 --- a/notebooks/template_discovery_brief.qmd +++ b/notebooks/template_discovery_brief.qmd @@ -148,8 +148,7 @@ print('Date and time: ' + str(datetime.datetime.now())) meta = pd.read_csv(sample_table, sep=',') -# alias_col is optional in the samplesheet; default to the sample name when -# absent so the alias-keyed columns/labels below always have a value. +# alias_col is optional; default to sample name if missing. if alias_col not in meta.columns: meta[alias_col] = meta['sample'] diff --git a/notebooks/template_giana.qmd b/notebooks/template_giana.qmd index 028b63c..a8721ea 100644 --- a/notebooks/template_giana.qmd +++ b/notebooks/template_giana.qmd @@ -38,8 +38,7 @@ warnings.filterwarnings( ## reading sample metadata meta = pd.read_csv(sample_table, sep=',') -# alias_col is optional in the samplesheet; default to the sample name when -# absent so the alias-keyed groupings/labels below always have a value. +# alias_col is optional; default to sample name if missing. if alias_col not in meta.columns: meta[alias_col] = meta['sample'] diff --git a/notebooks/template_gliph.qmd b/notebooks/template_gliph.qmd index cbba151..33924ab 100644 --- a/notebooks/template_gliph.qmd +++ b/notebooks/template_gliph.qmd @@ -52,8 +52,7 @@ if timepoint_order_col not in meta.columns: rank_map = {t: r for r, t in enumerate(sorted(unique_timepoints, key=str))} meta[timepoint_order_col] = meta[timepoint_col].map(rank_map) -# alias_col is optional in the samplesheet; default to the sample name when -# absent so the alias-keyed columns/labels below always have a value. +# alias_col is optional; default to sample name if missing. if alias_col not in meta.columns: meta[alias_col] = meta['sample'] diff --git a/notebooks/template_overlap.qmd b/notebooks/template_overlap.qmd index 8b882b3..304d790 100644 --- a/notebooks/template_overlap.qmd +++ b/notebooks/template_overlap.qmd @@ -53,8 +53,7 @@ import re ## Reading sample metadata meta = pd.read_csv(sample_table, sep=',') -# alias_col is optional in the samplesheet; default to the sample name when -# absent so the alias-keyed columns/labels below always have a value. +# alias_col is optional; default to sample name if missing. if alias_col not in meta.columns: meta[alias_col] = meta['sample'] diff --git a/notebooks/template_pheno_bulk.qmd b/notebooks/template_pheno_bulk.qmd index 48a9ff1..f1c7f72 100644 --- a/notebooks/template_pheno_bulk.qmd +++ b/notebooks/template_pheno_bulk.qmd @@ -61,8 +61,7 @@ warnings.filterwarnings( meta = pd.read_csv(sample_table, sep=',') meta.drop(columns=['file'], inplace=True) -# alias_col is optional in the samplesheet; default to the sample name when -# absent so the alias-keyed columns/labels below always have a value. +# alias_col is optional; default to sample name if missing. if alias_col not in meta.columns: meta[alias_col] = meta['sample'] diff --git a/notebooks/template_pheno_sc.qmd b/notebooks/template_pheno_sc.qmd index 5b5e687..5124430 100644 --- a/notebooks/template_pheno_sc.qmd +++ b/notebooks/template_pheno_sc.qmd @@ -50,8 +50,7 @@ import re meta = pd.read_csv(sample_table, sep=',') meta.drop(columns=['file'], inplace=True) -# alias_col is optional in the samplesheet; default to the sample name when -# absent so the alias-keyed columns/labels below always have a value. +# alias_col is optional; default to sample name if missing. if alias_col not in meta.columns: meta[alias_col] = meta['sample'] diff --git a/notebooks/template_qc.qmd b/notebooks/template_qc.qmd index 866183e..23fdb8e 100644 --- a/notebooks/template_qc.qmd +++ b/notebooks/template_qc.qmd @@ -137,8 +137,7 @@ if timepoint_order_col not in meta.columns: rank_map = {t: r for r, t in enumerate(sorted(unique_timepoints, key=str))} meta[timepoint_order_col] = meta[timepoint_col].map(rank_map) -# alias_col is optional in the samplesheet; default to the sample name when -# absent so hover labels and per-sample groupings below always have a value. +# alias_col is optional; default to sample name if missing. if alias_col not in meta.columns: meta[alias_col] = meta['sample'] diff --git a/notebooks/template_sample.qmd b/notebooks/template_sample.qmd index edc420b..1494efe 100644 --- a/notebooks/template_sample.qmd +++ b/notebooks/template_sample.qmd @@ -50,8 +50,7 @@ warnings.filterwarnings( meta = pd.read_csv(sample_table, sep=',') meta.drop(columns=['file'], inplace=True) -# alias_col is optional in the samplesheet; default to the sample name when -# absent so the alias-keyed columns/labels below always have a value. +# alias_col is optional; default to sample name if missing. if alias_col not in meta.columns: meta[alias_col] = meta['sample'] diff --git a/workflows/tcrtoolkit_bulk.nf b/workflows/tcrtoolkit_bulk.nf index f06ccbb..a13874b 100644 --- a/workflows/tcrtoolkit_bulk.nf +++ b/workflows/tcrtoolkit_bulk.nf @@ -23,12 +23,7 @@ workflow TCRTOOLKIT_BULK { println("Running TCRTOOLKIT_BULK workflow...") - // Construct levels list from the run_sample, run_compare, and run_patient parameters. - // Cirro/Nextflow CLI overrides land here as the string "false", not Boolean false - - // Groovy truthiness treats any non-empty String (including "false") as true, so a - // plain `if (params.run_sample)` would never actually disable a level. Route every - // flag through GDK's String.toBoolean() (via toString(), which also normalizes real - // Booleans and null) to parse the intended value correctly regardless of source. + // .toBoolean(): CLI overrides can arrive as the string "false", which is truthy in Groovy. def levels = [] if (params.run_sample?.toString()?.toBoolean() ?: false) levels << 'sample' if (params.run_compare?.toString()?.toBoolean() ?: false) levels << 'compare' From 9d98276f61b87ab5f8b192002407922afcadc6a0 Mon Sep 17 00:00:00 2001 From: dltamayo Date: Thu, 24 Sep 2026 10:27:05 -0400 Subject: [PATCH 09/19] Bump RENDER_NOTEBOOK to process_high - process_medium still OOMs Dataset a367a549 (fix/bulk-cirro-config with RENDER_NOTEBOOK already at process_medium/16GB) OOM-killed on template_details_sample, template_details_patient, template_discovery_brief, and template_details_compare - template_details_sample's kernel died mid- render (cell 20/23). These reports embed far more per-sample data than template_qc (which passed at 16GB): TCRdist3 HDF5 distance matrices, OLGA pgen tables, VDJDB match tables, convergence tables, all via embed-resources: true. Co-Authored-By: Claude Sonnet 5 --- conf/modules.config | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/conf/modules.config b/conf/modules.config index 3f5f39d..b8eae34 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -24,8 +24,8 @@ process { } withName: RENDER_NOTEBOOK { - // patient-details report OOMs at process_single's 2.GB default. - label = 'process_medium' + // details/discovery reports OOM even at process_medium's 16.GB. + label = 'process_high' publishDir = [ path: { "${params.outdir}/report" }, mode: params.publish_dir_mode, From 973bc8fc591d777a15fad328ab0a3716e4d36814 Mon Sep 17 00:00:00 2001 From: dltamayo Date: Thu, 24 Sep 2026 12:11:00 -0400 Subject: [PATCH 10/19] Bump RENDER_NOTEBOOK to process_high_memory (256GB) - still OOMs at 64GB Dataset fa684158 confirmed it synced to the latest commit (9d98276f61, already includes the process_high/64GB bump) and still OOM-killed on template_details_patient - but this time all 17 notebook cells completed successfully first; the OOM happened during pandoc's embed-resources HTML-embedding step, not Python execution. Points at the interactive Plotly figures (GIANA/GLIPH2 network + sunburst plots) being expensive for pandoc to self-contain, not the underlying computation. Add process_high_memory alongside process_high, matching the existing TCRDIST3_MATRIX pattern in this file, to get 256GB while keeping process_high's cpu/time settings. Co-Authored-By: Claude Sonnet 5 --- conf/modules.config | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/conf/modules.config b/conf/modules.config index b8eae34..76170d4 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -24,8 +24,8 @@ process { } withName: RENDER_NOTEBOOK { - // details/discovery reports OOM even at process_medium's 16.GB. - label = 'process_high' + // details/discovery reports OOM in pandoc's embed-resources step even at 64.GB. + label = ['process_high', 'process_high_memory'] publishDir = [ path: { "${params.outdir}/report" }, mode: params.publish_dir_mode, From 0385a0ce8031ce598bfe5716924749e7d0ff1ddc Mon Sep 17 00:00:00 2001 From: dltamayo Date: Thu, 24 Sep 2026 12:24:58 -0400 Subject: [PATCH 11/19] Vectorize expansion/contraction Fisher's exact tests template_overlap.qmd and template_discovery_brief.qmd (both hit by the same copy-pasted code) ran fisher_exact via comp_df.apply(axis=1) once per surviving clonotype per (subject, origin, timepoint-pair) group - likely the real cause of template_details_compare's 84min runtime and template_discovery_brief's 26min runtime before both OOM'd at 256GB, not an actual memory-sizing problem. total_pre/total_post are fixed margins within each group (assigned as a single scalar to the whole comp_df), so the one-sided Fisher's exact p-value reduces to a hypergeometric survival/CDF call, vectorizable across every row in the group at once instead of one Python call per clonotype. Verified the exact function as written in both files against scipy.stats.fisher_exact across random cases, zero counts, ties, extreme margin imbalance, and a 50k-row stress test - all identical to floating-point epsilon. ~60x faster in a synthetic 20k-row benchmark. Co-Authored-By: Claude Sonnet 5 --- notebooks/template_discovery_brief.qmd | 24 ++++++++++++++++-------- notebooks/template_overlap.qmd | 26 +++++++++++++++++--------- 2 files changed, 33 insertions(+), 17 deletions(-) diff --git a/notebooks/template_discovery_brief.qmd b/notebooks/template_discovery_brief.qmd index 8bb7895..d459e07 100644 --- a/notebooks/template_discovery_brief.qmd +++ b/notebooks/template_discovery_brief.qmd @@ -896,7 +896,7 @@ def create_styled_table(df, title="", max_height='400px', formatter=None): import pandas as pd import numpy as np import itertools -from scipy.stats import fisher_exact +from scipy.stats import hypergeom from statsmodels.stats.multitest import multipletests # --- Setup & Pre-computation --- @@ -915,11 +915,15 @@ sample_total_counts.rename(columns={'duplicate_count': 'total_counts'}, inplace= clonotypes_df = pd.merge(clonotypes_df, sample_total_counts, on=[subject_col, 'origin', timepoint_col]) clonotypes_df['frequency'] = clonotypes_df['duplicate_count'] / clonotypes_df['total_counts'] -# Helper to run fisher exact test on rows -def run_fisher(row, alt_hyp): - table = [[row['count_post'], row['total_post'] - row['count_post']], - [row['count_pre'], row['total_pre'] - row['count_pre']]] - return fisher_exact(table, alternative=alt_hyp)[1] +# Vectorized one-sided Fisher's exact test: total_pre/total_post are fixed +# margins within a group, so this reduces to a single hypergeometric call +# per group instead of one fisher_exact call per clonotype. +def run_fisher_vectorized(count_post, count_pre, total_post, total_pre, alt_hyp): + M = total_pre + total_post + n = count_post + count_pre + if alt_hyp == 'greater': + return hypergeom.sf(count_post - 1, M, n, total_post) + return hypergeom.cdf(count_post, M, n, total_post) # ========================================== @@ -972,7 +976,9 @@ for (subject, origin), subject_df in clonotypes_df.groupby([subject_col, 'origin continue # 5. Apply Fisher's Exact test only to the remaining rows - comp_df['p_value'] = comp_df.apply(run_fisher, axis=1, alt_hyp='greater') + comp_df['p_value'] = run_fisher_vectorized( + comp_df['count_post'], comp_df['count_pre'], comp_df['total_post'], comp_df['total_pre'], 'greater' + ) # Add metadata comp_df[subject_col] = subject @@ -1094,7 +1100,9 @@ for (subject, origin), subject_df in clonotypes_df.groupby([subject_col, 'origin continue # 5. Apply Fisher - comp_df['p_value'] = comp_df.apply(run_fisher, axis=1, alt_hyp='less') + comp_df['p_value'] = run_fisher_vectorized( + comp_df['count_post'], comp_df['count_pre'], comp_df['total_post'], comp_df['total_pre'], 'less' + ) # Add metadata comp_df[subject_col] = subject diff --git a/notebooks/template_overlap.qmd b/notebooks/template_overlap.qmd index 304d790..62f377d 100644 --- a/notebooks/template_overlap.qmd +++ b/notebooks/template_overlap.qmd @@ -23,7 +23,7 @@ concat_csv = f"{project_dir}/annotate/concatenated_cdr3_sorted.tsv" from IPython.display import Image, display, Markdown, HTML from matplotlib.colors import LinearSegmentedColormap from scipy.sparse import csr_matrix -from scipy.stats import gaussian_kde, fisher_exact +from scipy.stats import gaussian_kde from statsmodels.stats.multitest import multipletests from scipy.cluster.hierarchy import linkage, leaves_list from scipy.spatial.distance import pdist @@ -153,7 +153,7 @@ def create_styled_table(df, title="", max_height='400px', formatter=None): import pandas as pd import numpy as np import itertools -from scipy.stats import fisher_exact +from scipy.stats import hypergeom from statsmodels.stats.multitest import multipletests # --- Setup & Pre-computation --- @@ -172,11 +172,15 @@ sample_total_counts.rename(columns={'duplicate_count': 'total_counts'}, inplace= clonotypes_df = pd.merge(clonotypes_df, sample_total_counts, on=[subject_col, 'origin', timepoint_col]) clonotypes_df['frequency'] = clonotypes_df['duplicate_count'] / clonotypes_df['total_counts'] -# Helper to run fisher exact test on rows -def run_fisher(row, alt_hyp): - table = [[row['count_post'], row['total_post'] - row['count_post']], - [row['count_pre'], row['total_pre'] - row['count_pre']]] - return fisher_exact(table, alternative=alt_hyp)[1] +# Vectorized one-sided Fisher's exact test: total_pre/total_post are fixed +# margins within a group, so this reduces to a single hypergeometric call +# per group instead of one fisher_exact call per clonotype. +def run_fisher_vectorized(count_post, count_pre, total_post, total_pre, alt_hyp): + M = total_pre + total_post + n = count_post + count_pre + if alt_hyp == 'greater': + return hypergeom.sf(count_post - 1, M, n, total_post) + return hypergeom.cdf(count_post, M, n, total_post) # ========================================== @@ -229,7 +233,9 @@ for (subject, origin), subject_df in clonotypes_df.groupby([subject_col, 'origin continue # 5. Apply Fisher's Exact test only to the remaining rows - comp_df['p_value'] = comp_df.apply(run_fisher, axis=1, alt_hyp='greater') + comp_df['p_value'] = run_fisher_vectorized( + comp_df['count_post'], comp_df['count_pre'], comp_df['total_post'], comp_df['total_pre'], 'greater' + ) # Add metadata comp_df[subject_col] = subject @@ -351,7 +357,9 @@ for (subject, origin), subject_df in clonotypes_df.groupby([subject_col, 'origin continue # 5. Apply Fisher - comp_df['p_value'] = comp_df.apply(run_fisher, axis=1, alt_hyp='less') + comp_df['p_value'] = run_fisher_vectorized( + comp_df['count_post'], comp_df['count_pre'], comp_df['total_post'], comp_df['total_pre'], 'less' + ) # Add metadata comp_df[subject_col] = subject From 65ef10b2ca94b9a3107c182d767037a32e0a100d Mon Sep 17 00:00:00 2001 From: dltamayo Date: Thu, 24 Sep 2026 14:59:08 -0400 Subject: [PATCH 12/19] Fix RENDER_NOTEBOOK's memory bump never actually applying The last few commits bumped RENDER_NOTEBOOK's memory via `withName: RENDER_NOTEBOOK { label = [...] }` in conf/modules.config, copying TCRDIST3_MATRIX's pattern - but base.config (with all the withLabel:process_* blocks) is includeConfig'd before modules.config. Nextflow resolves config in one top-to-bottom pass: RENDER_NOTEBOOK's inline `label 'process_single'` already matched `withLabel:process_single { memory = 2.GB }` by the time base.config loaded, and reassigning the label later in modules.config never gave the (earlier-declared) process_high/process_high_memory withLabel blocks a second chance to fire. Confirmed with an isolated repro using the actual withLabel blocks: task.memory stayed 2 GB despite the override. This is why the workflow report kept showing every RENDER_NOTEBOOK task at 1 cpu / 2 GB regardless of the last two "bump to process_high(_memory)" commits. modules/local/compare/gliph2.nf already has the correct pattern for a process that needs multiple resource-tier labels: declare them all inline in the process definition, not via a later config override, since inline labels are known before any config file parses and every matching withLabel block applies. Move RENDER_NOTEBOOK's labels there instead. Verified via a faithful isolated repro (byte-identical withLabel blocks): inline multi-label resolves to 256GB; the config-override version does not. Note: TCRDIST3_MATRIX's dense-matrix branch (`matrix_sparsity != 'sparse'`) uses the same config-override pattern this commit moves away from, and is likely subject to the identical bug (silently capped at process_high's 64GB instead of process_high_memory's 256GB) - pre-existing, not introduced here, not fixed in this commit. Co-Authored-By: Claude Sonnet 5 --- conf/modules.config | 2 -- modules/local/report/render_notebook.nf | 3 ++- 2 files changed, 2 insertions(+), 3 deletions(-) diff --git a/conf/modules.config b/conf/modules.config index 76170d4..1b959dd 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -24,8 +24,6 @@ process { } withName: RENDER_NOTEBOOK { - // details/discovery reports OOM in pandoc's embed-resources step even at 64.GB. - label = ['process_high', 'process_high_memory'] publishDir = [ path: { "${params.outdir}/report" }, mode: params.publish_dir_mode, diff --git a/modules/local/report/render_notebook.nf b/modules/local/report/render_notebook.nf index a25ac23..5ebf8ab 100644 --- a/modules/local/report/render_notebook.nf +++ b/modules/local/report/render_notebook.nf @@ -1,7 +1,8 @@ // Generic process to render a Quarto notebook to HTML process RENDER_NOTEBOOK { tag "${notebook.getBaseName()}" - label 'process_single' + label 'process_high' + label 'process_high_memory' input: // path(files) stages files flat in the root dir; staged_layout optionally From 3c5aa462ccf5077d77776c82d7ccbd974ae19d5b Mon Sep 17 00:00:00 2001 From: dltamayo Date: Tue, 29 Sep 2026 13:06:00 -0400 Subject: [PATCH 13/19] Fix template_discovery_brief's VDJdb section for airr-native input Dataset af639006 got through every other report (qc, details_patient, details_sample, details_compare - the last two confirming the vectorized Fisher's exact fix works) and failed on template_discovery_brief: Warning: File s3://.../data/convert/su005_BCC_post_TCRB_airr.tsv not found. Skipping. (x4) ValueError: No objects to concatenate The VDJdb section falls back to `row['file']` when no adaptive conversion ran (input_format == 'airr'), but that's the raw samplesheet value - an S3 URI in this pipeline, never a locally readable path. workflows/tcrtoolkit_bulk.nf deliberately left convert_files empty for the non-adaptive branch, so nothing was ever staged locally for this notebook to read regardless of what's in that samplesheet column. Populate convert_files from INPUT_CHECK's sample_map (not just CONVERT's output) when input_format != 'adaptive', carrying [meta, file] pairs through so bulktcr_analysis.nf can stage each one under ${meta.sample}_airr.tsv - the name the notebook looks for - regardless of the source file's actual basename. This isn't specific to files that happen to already end in "_airr.tsv" (true of this test dataset only because it was itself pre-converted); it handles arbitrarily-named native AIRR input the same way. Verified against the actual notebook code (copied verbatim into a harness): reproduced the original ValueError with an S3-URI samplesheet and empty convert_dir, then confirmed it succeeds once convert_dir is populated per this fix - using deliberately unrelated source filenames to prove the fix doesn't depend on any naming coincidence. Co-Authored-By: Claude Sonnet 5 --- subworkflows/local/bulktcr_analysis.nf | 7 +++++-- workflows/tcrtoolkit_bulk.nf | 9 ++++----- 2 files changed, 9 insertions(+), 7 deletions(-) diff --git a/subworkflows/local/bulktcr_analysis.nf b/subworkflows/local/bulktcr_analysis.nf index 857ba17..e016205 100644 --- a/subworkflows/local/bulktcr_analysis.nf +++ b/subworkflows/local/bulktcr_analysis.nf @@ -165,8 +165,11 @@ workflow BULKTCR_ANALYSIS { .combine(ch_tcrpheno_files.map { l -> [l] }) .combine(pseudobulk_pheno_files.map { l -> [l] }) .map { sample_stats_csv, concat_cdr3_sorted, shared_cdr3_file, tcrdist_files_l, vdjdb_files_l, convert_files_l, tcrpheno_files_l, pseudobulk_files_l -> + // convert_files_l is [meta, file] pairs - staged as ${meta.sample}_airr.tsv + // regardless of the source's original basename, so the notebook can find it + // whether it came from CONVERT_ADAPTIVE or is native (already-AIRR) input. def report_files = [sample_stats_csv, concat_cdr3_sorted, shared_cdr3_file, pheno_notebook] + - tcrdist_files_l + vdjdb_files_l + convert_files_l + tcrpheno_files_l + pseudobulk_files_l + tcrdist_files_l + vdjdb_files_l + convert_files_l.collect { _meta, f -> f } + tcrpheno_files_l + pseudobulk_files_l def staged_layout = [ ["${params.project_name}/sample/${sample_stats_csv.name}", sample_stats_csv.name], ["${params.project_name}/annotate/${concat_cdr3_sorted.name}", concat_cdr3_sorted.name], @@ -174,7 +177,7 @@ workflow BULKTCR_ANALYSIS { ["template_pheno.qmd", pheno_notebook.name] ] + tcrdist_files_l.collect { f -> ["${params.project_name}/tcrdist3/${f.name}", f.name] } + vdjdb_files_l.collect { f -> ["${params.project_name}/vdjdb/${f.name}", f.name] } + - convert_files_l.collect { f -> ["${params.project_name}/convert/${f.name}", f.name] } + + convert_files_l.collect { meta, f -> ["${params.project_name}/convert/${meta.sample}_airr.tsv", f.name] } + tcrpheno_files_l.collect { f -> ["${params.project_name}/tcrpheno/${f.name}", f.name] } + pseudobulk_files_l.collect { f -> ["${params.project_name}/pseudobulk/${f.name}", f.name] } tuple( diff --git a/workflows/tcrtoolkit_bulk.nf b/workflows/tcrtoolkit_bulk.nf index a13874b..b99fb45 100644 --- a/workflows/tcrtoolkit_bulk.nf +++ b/workflows/tcrtoolkit_bulk.nf @@ -65,12 +65,11 @@ workflow TCRTOOLKIT_BULK { sample_map_final = INPUT_CHECK.out.sample_map } - // template_discovery_brief.qmd stages AIRR-converted files only when CONVERT ran - // (adaptive); template_discovery_brief.qmd's VDJdb section otherwise reads the raw - // input directly, which already has AIRR-standard frequency columns. + // [meta, file] pairs staged for template_discovery_brief.qmd's VDJdb section: the + // adaptive-converted output when CONVERT ran, otherwise the raw (already-AIRR) input. def convert_files = (input_format == 'adaptive') - ? CONVERT.out.map { _meta, f -> f }.collect() - : channel.value([]) + ? CONVERT.out.collect(flat: false) + : INPUT_CHECK.out.sample_map.collect(flat: false) // Bulk reports are sample-centric. Compare- and patient-dependent sections are // added only when those workflow levels are present. Change this once reports do From 4ee93c2b1be071f6db9fc5ca46ddcc5724fafdde Mon Sep 17 00:00:00 2001 From: dltamayo Date: Tue, 29 Sep 2026 16:31:32 -0400 Subject: [PATCH 14/19] Fix public-clones UpSet plot crash on single-patient datasets Dataset 70266df6 got past the convert_dir fix (no more "not found" warnings) and crashed later in template_discovery_brief: AttributeError: 'Index' object has no attribute 'levels' (upsetplot/reformat.py:71, _check_index) plot_public_clones_upset groups by subject_col (patient) to build the UpSet plot's intersection data. This test dataset has exactly one patient (su005), so that groupby produces a single-key dict; upsetplot.from_contents() returns a plain Index instead of a MultiIndex for a single category, and UpSet()/.plot() then raises AttributeError trying to access .levels on it. The existing guard at this call site (`except TypeError as e: ... "This typically happens if only one intersection group remains"`) already anticipated this failure mode by description, but wasn't catching the exception type this upsetplot version actually raises for it, so nothing caught it. Multi-patient datasets never hit this - groupby(subject_col) always yields >=2 groups - which is presumably why this never surfaced before. Add an explicit check for fewer than 2 patients and skip with a message, same as the existing `if upset_data.empty` guard just above it, instead of relying on catching whatever exception the library happens to raise. Verified against the actual function (extracted from the file): single-patient input now skips cleanly instead of crashing; multi-patient input still renders as before. Co-Authored-By: Claude Sonnet 5 --- notebooks/template_discovery_brief.qmd | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/notebooks/template_discovery_brief.qmd b/notebooks/template_discovery_brief.qmd index d459e07..e136fe4 100644 --- a/notebooks/template_discovery_brief.qmd +++ b/notebooks/template_discovery_brief.qmd @@ -2492,7 +2492,11 @@ def plot_public_clones_upset(concat_df, subject_col, min_shared_patients=2): # --- 1. Extract Unique Clones per Patient --- patient_clones = df_clean.groupby(subject_col)['junction_aa'].unique().to_dict() patient_contents = {str(patient): list(clones) for patient, clones in patient_clones.items()} - + + if len(patient_contents) < 2: + print("Only one patient present; skipping UpSet plot (needs ≥ 2 patients to show intersections).") + return + # --- 2. Convert to UpSet Multi-Index Format --- upset_data = from_contents(patient_contents) From a98a69cd64403457bcaee5813cc01ff8ca765823 Mon Sep 17 00:00:00 2001 From: dltamayo Date: Wed, 30 Sep 2026 10:10:10 -0400 Subject: [PATCH 15/19] Vectorize remaining .apply(axis=1) hotspots in overlap/discovery_brief notebooks template_details_compare (via template_overlap.qmd) and template_discovery_brief were the two most expensive report steps by cost/duration even after the earlier Fisher's-exact vectorization, because both still ran several row-wise .apply(axis=1)/.apply() calls, each of which builds a pandas Series per row. - template_overlap.qmd: row-wise z-score (`matrix_df.apply(zscore, axis=1, ...)`) replaced with `sub`/`div` against `mean`/`std(ddof=0)`; verified bit-for-bit equivalent (max abs diff: 0.0) against scipy.stats.zscore per-row. - template_overlap.qmd: `get_status` .apply(axis=1) replaced with a list comprehension over zip(); verified identical output against the original. - template_discovery_brief.qmd: `assign_status` .apply(axis=1) replaced the same way; verified identical output. - template_discovery_brief.qmd: three node_properties .apply()/.apply(axis=1) calls (color/category assignment, size, hover title) in the per-patient network-graph section replaced with list comprehensions / vectorized numpy; verified identical output (color, category, size, and title all match). All python code blocks in both files compile cleanly after the edits. Co-Authored-By: Claude Sonnet 5 --- notebooks/template_discovery_brief.qmd | 40 +++++++++++++------------- notebooks/template_overlap.qmd | 25 +++++++--------- 2 files changed, 30 insertions(+), 35 deletions(-) diff --git a/notebooks/template_discovery_brief.qmd b/notebooks/template_discovery_brief.qmd index e136fe4..3f2f3b7 100644 --- a/notebooks/template_discovery_brief.qmd +++ b/notebooks/template_discovery_brief.qmd @@ -1664,17 +1664,14 @@ for subject, subj_df in clonotypes_df.groupby(subject_col): pivot['freq_t1_plot'] = pivot[t1].replace(0, min_freq) pivot['freq_t2_plot'] = pivot[t2].replace(0, min_freq) - # Assign highlight status using our sets - def assign_status(row): - key = (subject, origin, t1, t2, row['junction_aa']) - if key in sig_exp_lookup: - return "Expanded" - elif key in sig_cont_lookup: - return "Contracted" - else: - return "Not Significant" - - pivot['Status'] = pivot.apply(assign_status, axis=1) + # Assign highlight status using our sets. List comprehension over the + # junction_aa column avoids the per-row Series-construction overhead + # of .apply(axis=1); logic is identical to the row-wise version. + keys = [(subject, origin, t1, t2, j) for j in pivot['junction_aa']] + pivot['Status'] = [ + "Expanded" if key in sig_exp_lookup else ("Contracted" if key in sig_cont_lookup else "Not Significant") + for key in keys + ] # Sort so that colored points render on top of gray points pivot['sort_order'] = pivot['Status'].map({"Not Significant": 0, "Expanded": 1, "Contracted": 2}) @@ -1881,24 +1878,27 @@ else: if len(tps) > 1: return COLOR_MULTI, 'Multiple Timepoints' else: return timepoint_colors.get(tps[0], '#888888'), tps[0] - node_properties[['color', 'category']] = node_properties['timepoints'].apply( - lambda x: pd.Series(get_color_and_category(x)) - ) + colors_categories = [get_color_and_category(x) for x in node_properties['timepoints']] + node_properties['color'] = [c for c, _ in colors_categories] + node_properties['category'] = [cat for _, cat in colors_categories] - node_properties['size'] = node_properties['max_frequency'].apply(lambda x: max(10, np.sqrt(x * 100) * 15)) + node_properties['size'] = np.maximum(10, np.sqrt(node_properties['max_frequency'] * 100) * 15) tcr_to_idx = {tcr: i for i, tcr in enumerate(node_properties['cdr3_b_aa'])} g = ig.Graph() g.add_vertices(len(node_properties)) g.vs['name'] = node_properties['cdr3_b_aa'] - g.vs['size'] = node_properties['size'] + g.vs['size'] = node_properties['size'] g.vs['color'] = node_properties['color'] g.vs['category'] = node_properties['category'] - g.vs['title'] = node_properties.apply( - lambda row: f"TCR: {row['cdr3_b_aa']}
Peak Freq: {(row['max_frequency']*100):.3f}%
Peak Count: {row['max_count']}
TPs: {row['timepoints']}", - axis=1 - ).tolist() + g.vs['title'] = [ + f"TCR: {cdr3}
Peak Freq: {freq*100:.3f}%
Peak Count: {count}
TPs: {tps}" + for cdr3, freq, count, tps in zip( + node_properties['cdr3_b_aa'], node_properties['max_frequency'], + node_properties['max_count'], node_properties['timepoints'] + ) + ] patient_samples = patient_df['sample'].unique() for sample_id in patient_samples: diff --git a/notebooks/template_overlap.qmd b/notebooks/template_overlap.qmd index 62f377d..c1b476d 100644 --- a/notebooks/template_overlap.qmd +++ b/notebooks/template_overlap.qmd @@ -451,7 +451,6 @@ The heatmap displays only those TCR clonotypes that were identified as **"signif import plotly.express as px import plotly.graph_objects as go import scipy.cluster.hierarchy as sch -from scipy.stats import zscore from IPython.display import display, Markdown import pandas as pd @@ -536,7 +535,7 @@ else: # --- Calculate Z-Scores --- if matrix_df.shape[1] > 1: # fillna(0) handles cases where standard deviation is 0 (identical frequencies across time) - z_score_df = matrix_df.apply(zscore, axis=1, result_type='expand').fillna(0) + z_score_df = matrix_df.sub(matrix_df.mean(axis=1), axis=0).div(matrix_df.std(axis=1, ddof=0), axis=0).fillna(0) else: z_score_df = matrix_df.copy() * 0 # If only 1 timepoint, z-score is baseline 0 @@ -783,21 +782,17 @@ def generate_plot_quarto_tabs(id_map, title_prefix, line_color): print("*No clonotypes met the significance thresholds for this direction; " "showing background repertoire trajectories only.*\n") - # Define Status Helper - def get_status(row): - subj = row[subject_col] - orig = row['origin'] if 'origin' in row else None - seq = row['junction_aa'] - - # Check map using the correct key format - key = (subj, orig) if orig is not None else subj - if key in id_map and seq in id_map[key]: - return "Highlight" - return "Background" - # Create temp dataframe plot_data = cdr3_df.copy() - plot_data['status'] = plot_data.apply(get_status, axis=1) + + # List comprehension over zip() avoids the per-row Series-construction + # overhead of .apply(axis=1); logic is identical to the row-wise version. + has_origin = 'origin' in plot_data.columns + keys = list(zip(plot_data[subject_col], plot_data['origin'])) if has_origin else plot_data[subject_col].tolist() + plot_data['status'] = [ + "Highlight" if (key in id_map and seq in id_map[key]) else "Background" + for key, seq in zip(keys, plot_data['junction_aa']) + ] subjects = sorted(plot_data[subject_col].unique()) color_map = {"Highlight": line_color, "Background": "lightgrey"} From a30f79039734ad809abaacb5cc208b18cfb48dd9 Mon Sep 17 00:00:00 2001 From: dltamayo Date: Wed, 30 Sep 2026 10:28:17 -0400 Subject: [PATCH 16/19] RENDER_NOTEBOOK: scale resources per-notebook instead of a flat 16cpu/256GB All five report notebooks shared the same process_high + process_high_memory labels (16 cpu, 256GB, 16h), even though the last completed real run (ac1f14c8) showed durations ranging from 30s (template_qc) to 117 minutes (template_discovery_brief) - the three fast notebooks (template_qc, template_details_patient, template_details_sample) finish in under 5 minutes and don't touch the large VDJdb match files or build network graphs. Replaced the static labels with dynamic cpus/memory/time directives (same pattern already used in modules/local/sample/tcrdist3.nf) keyed on notebook.baseName, so only template_details_compare and template_discovery_brief keep the 16cpu/256GB tier; everything else drops to 4cpu/16GB/4h. Verified locally with -stub-run: template_qc resolves to the light tier and completes on a 10-cpu machine; template_discovery_brief resolves to the heavy tier's 16-cpu request (confirmed by it correctly exceeding the 10-cpu local executor's capacity). All 3 render_notebook.nf.test cases still pass. The two heavy notebooks are left at 256GB pending real peak-memory numbers from the Nextflow execution report, since the vectorization fixes in this branch may allow lowering that tier further. Co-Authored-By: Claude Sonnet 5 --- modules/local/report/render_notebook.nf | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/modules/local/report/render_notebook.nf b/modules/local/report/render_notebook.nf index 5ebf8ab..e7589a0 100644 --- a/modules/local/report/render_notebook.nf +++ b/modules/local/report/render_notebook.nf @@ -1,8 +1,13 @@ // Generic process to render a Quarto notebook to HTML process RENDER_NOTEBOOK { tag "${notebook.getBaseName()}" - label 'process_high' - label 'process_high_memory' + + // template_details_compare and template_discovery_brief process full VDJdb match + // files (up to ~1GB/sample) and build per-patient network graphs; the rest + // aggregate small summary tables and finish in under 5 minutes. + cpus { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16 * task.attempt : 4 * task.attempt } + memory { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 256.GB * task.attempt : 16.GB * task.attempt } + time { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16.h : 4.h } input: // path(files) stages files flat in the root dir; staged_layout optionally From 89d4f616c50e33853b84b885cd5b19abbdf5ea07 Mon Sep 17 00:00:00 2001 From: dltamayo Date: Wed, 30 Sep 2026 10:49:15 -0400 Subject: [PATCH 17/19] RENDER_NOTEBOOK: move resource tiering into conf/modules.config, shrink heavy tier Per-notebook resource directives previously lived inline in render_notebook.nf. Moved them into conf/modules.config's existing withName: RENDER_NOTEBOOK block instead, keeping the process definition focused on rendering logic. This is safe unlike the earlier RENDER_NOTEBOOK OOM bug (config-level `label = [...]` reassignments never triggering the matching withLabel: block due to include ordering) because withName: here sets cpus/memory/time directly rather than routing through a label - verified locally against the real nextflow.config chain (-stub-run): the closures correctly read the notebook input variable and resolve template_qc to 4cpu/16GB and the two heavy notebooks to 16cpu/64GB. Also lowered the heavy-notebook memory tier from process_high_memory (256GB) to plain process_high (64GB): the Nextflow execution report for the last completed real run (dataset ac1f14c8) shows template_details_compare peaked at 5.1% of 256GB (~13GB) and template_discovery_brief at 0.2% (~0.5GB). The 256GB figure was never actually validated against real usage - it came from iteratively doubling memory while chasing the label-ordering bug above, which was the real blocker the whole time. 64GB keeps ~5x headroom over the observed peak for the larger of the two. All 3 render_notebook.nf.test cases still pass. Co-Authored-By: Claude Sonnet 5 --- conf/modules.config | 9 +++++++++ modules/local/report/render_notebook.nf | 7 ------- 2 files changed, 9 insertions(+), 7 deletions(-) diff --git a/conf/modules.config b/conf/modules.config index 1b959dd..fe3a9e3 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -29,6 +29,15 @@ process { mode: params.publish_dir_mode, saveAs: { filename -> filename.equals('versions.yml') ? null : filename } ] + // template_details_compare and template_discovery_brief process full VDJdb + // match files (up to ~1GB/sample) and build per-patient network graphs; the + // rest aggregate small summary tables and finish in under 5 minutes. Peak + // memory measured on a real run (dataset ac1f14c8): compare ~13GB (5.1% of + // 256GB), discovery_brief ~0.5GB (0.2%) - 64GB keeps ~5x headroom over the + // observed peak without the earlier blind 256GB allocation. + cpus = { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16 * task.attempt : 4 * task.attempt } + memory = { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 64.GB * task.attempt : 16.GB * task.attempt } + time = { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16.h : 4.h } } } \ No newline at end of file diff --git a/modules/local/report/render_notebook.nf b/modules/local/report/render_notebook.nf index e7589a0..5717d7d 100644 --- a/modules/local/report/render_notebook.nf +++ b/modules/local/report/render_notebook.nf @@ -2,13 +2,6 @@ process RENDER_NOTEBOOK { tag "${notebook.getBaseName()}" - // template_details_compare and template_discovery_brief process full VDJdb match - // files (up to ~1GB/sample) and build per-patient network graphs; the rest - // aggregate small summary tables and finish in under 5 minutes. - cpus { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16 * task.attempt : 4 * task.attempt } - memory { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 256.GB * task.attempt : 16.GB * task.attempt } - time { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16.h : 4.h } - input: // path(files) stages files flat in the root dir; staged_layout optionally // symlinks them into a project_dir-style subdirectory tree for notebooks that From 01152376f14d7c2fdf5ddd45732130706ff077ac Mon Sep 17 00:00:00 2001 From: dltamayo Date: Wed, 30 Sep 2026 10:53:58 -0400 Subject: [PATCH 18/19] Trim multi-line comments added this session down to one line each Longer rationale belongs in commit messages, not inline. No functional changes - comment text only. Co-Authored-By: Claude Sonnet 5 --- conf/modules.config | 7 +------ notebooks/template_discovery_brief.qmd | 8 ++------ notebooks/template_overlap.qmd | 7 ++----- subworkflows/local/bulktcr_analysis.nf | 4 +--- workflows/tcrtoolkit_bulk.nf | 3 +-- 5 files changed, 7 insertions(+), 22 deletions(-) diff --git a/conf/modules.config b/conf/modules.config index fe3a9e3..d72982a 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -29,12 +29,7 @@ process { mode: params.publish_dir_mode, saveAs: { filename -> filename.equals('versions.yml') ? null : filename } ] - // template_details_compare and template_discovery_brief process full VDJdb - // match files (up to ~1GB/sample) and build per-patient network graphs; the - // rest aggregate small summary tables and finish in under 5 minutes. Peak - // memory measured on a real run (dataset ac1f14c8): compare ~13GB (5.1% of - // 256GB), discovery_brief ~0.5GB (0.2%) - 64GB keeps ~5x headroom over the - // observed peak without the earlier blind 256GB allocation. + // template_details_compare and template_discovery_brief need far more cpu/memory than the other notebooks. cpus = { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16 * task.attempt : 4 * task.attempt } memory = { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 64.GB * task.attempt : 16.GB * task.attempt } time = { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16.h : 4.h } diff --git a/notebooks/template_discovery_brief.qmd b/notebooks/template_discovery_brief.qmd index 3f2f3b7..178494c 100644 --- a/notebooks/template_discovery_brief.qmd +++ b/notebooks/template_discovery_brief.qmd @@ -915,9 +915,7 @@ sample_total_counts.rename(columns={'duplicate_count': 'total_counts'}, inplace= clonotypes_df = pd.merge(clonotypes_df, sample_total_counts, on=[subject_col, 'origin', timepoint_col]) clonotypes_df['frequency'] = clonotypes_df['duplicate_count'] / clonotypes_df['total_counts'] -# Vectorized one-sided Fisher's exact test: total_pre/total_post are fixed -# margins within a group, so this reduces to a single hypergeometric call -# per group instead of one fisher_exact call per clonotype. +# One-sided Fisher's exact test reduces to a single vectorized hypergeometric call since total_pre/total_post are fixed margins. def run_fisher_vectorized(count_post, count_pre, total_post, total_pre, alt_hyp): M = total_pre + total_post n = count_post + count_pre @@ -1664,9 +1662,7 @@ for subject, subj_df in clonotypes_df.groupby(subject_col): pivot['freq_t1_plot'] = pivot[t1].replace(0, min_freq) pivot['freq_t2_plot'] = pivot[t2].replace(0, min_freq) - # Assign highlight status using our sets. List comprehension over the - # junction_aa column avoids the per-row Series-construction overhead - # of .apply(axis=1); logic is identical to the row-wise version. + # Avoids the per-row Series-construction overhead of .apply(axis=1). keys = [(subject, origin, t1, t2, j) for j in pivot['junction_aa']] pivot['Status'] = [ "Expanded" if key in sig_exp_lookup else ("Contracted" if key in sig_cont_lookup else "Not Significant") diff --git a/notebooks/template_overlap.qmd b/notebooks/template_overlap.qmd index c1b476d..ae18596 100644 --- a/notebooks/template_overlap.qmd +++ b/notebooks/template_overlap.qmd @@ -172,9 +172,7 @@ sample_total_counts.rename(columns={'duplicate_count': 'total_counts'}, inplace= clonotypes_df = pd.merge(clonotypes_df, sample_total_counts, on=[subject_col, 'origin', timepoint_col]) clonotypes_df['frequency'] = clonotypes_df['duplicate_count'] / clonotypes_df['total_counts'] -# Vectorized one-sided Fisher's exact test: total_pre/total_post are fixed -# margins within a group, so this reduces to a single hypergeometric call -# per group instead of one fisher_exact call per clonotype. +# One-sided Fisher's exact test reduces to a single vectorized hypergeometric call since total_pre/total_post are fixed margins. def run_fisher_vectorized(count_post, count_pre, total_post, total_pre, alt_hyp): M = total_pre + total_post n = count_post + count_pre @@ -785,8 +783,7 @@ def generate_plot_quarto_tabs(id_map, title_prefix, line_color): # Create temp dataframe plot_data = cdr3_df.copy() - # List comprehension over zip() avoids the per-row Series-construction - # overhead of .apply(axis=1); logic is identical to the row-wise version. + # Avoids the per-row Series-construction overhead of .apply(axis=1). has_origin = 'origin' in plot_data.columns keys = list(zip(plot_data[subject_col], plot_data['origin'])) if has_origin else plot_data[subject_col].tolist() plot_data['status'] = [ diff --git a/subworkflows/local/bulktcr_analysis.nf b/subworkflows/local/bulktcr_analysis.nf index e016205..6c5fa88 100644 --- a/subworkflows/local/bulktcr_analysis.nf +++ b/subworkflows/local/bulktcr_analysis.nf @@ -165,9 +165,7 @@ workflow BULKTCR_ANALYSIS { .combine(ch_tcrpheno_files.map { l -> [l] }) .combine(pseudobulk_pheno_files.map { l -> [l] }) .map { sample_stats_csv, concat_cdr3_sorted, shared_cdr3_file, tcrdist_files_l, vdjdb_files_l, convert_files_l, tcrpheno_files_l, pseudobulk_files_l -> - // convert_files_l is [meta, file] pairs - staged as ${meta.sample}_airr.tsv - // regardless of the source's original basename, so the notebook can find it - // whether it came from CONVERT_ADAPTIVE or is native (already-AIRR) input. + // convert_files_l is [meta, file] pairs, staged as ${meta.sample}_airr.tsv regardless of source basename. def report_files = [sample_stats_csv, concat_cdr3_sorted, shared_cdr3_file, pheno_notebook] + tcrdist_files_l + vdjdb_files_l + convert_files_l.collect { _meta, f -> f } + tcrpheno_files_l + pseudobulk_files_l def staged_layout = [ diff --git a/workflows/tcrtoolkit_bulk.nf b/workflows/tcrtoolkit_bulk.nf index b99fb45..df72205 100644 --- a/workflows/tcrtoolkit_bulk.nf +++ b/workflows/tcrtoolkit_bulk.nf @@ -65,8 +65,7 @@ workflow TCRTOOLKIT_BULK { sample_map_final = INPUT_CHECK.out.sample_map } - // [meta, file] pairs staged for template_discovery_brief.qmd's VDJdb section: the - // adaptive-converted output when CONVERT ran, otherwise the raw (already-AIRR) input. + // [meta, file] pairs for template_discovery_brief.qmd's VDJdb section. def convert_files = (input_format == 'adaptive') ? CONVERT.out.collect(flat: false) : INPUT_CHECK.out.sample_map.collect(flat: false) From a7e137b1cb33c98afdb08d47108aaa5027bbfca7 Mon Sep 17 00:00:00 2001 From: dltamayo Date: Wed, 30 Sep 2026 17:37:46 -0400 Subject: [PATCH 19/19] RENDER_NOTEBOOK: move template_details_sample into the 64GB tier Dataset 52249ba1 failed: template_details_sample OOM'd at cell 20/23 under the 16GB "light" tier assigned based on its short (4m20s) duration in the prior run. Duration was the wrong proxy - template_sample.qmd (included by template_details_sample.qmd) loads each sample's TCRdist distance matrix from HDF5 and converts it to a dense array via sparse_matrix.toarray(), then symmetrizes it with np.maximum(dense, dense.T), holding two full NxN float64 arrays at once. For su005's ~25-30k clones per sample that's a multi-GB memory spike that doesn't show up in wall-clock time. template_qc and template_details_patient both completed fine at 16GB in this same run, so they stay on the light tier. template_details_sample moves to the same 64GB tier already verified for template_details_compare/ template_discovery_brief, rather than guessing a new number. Verified locally against the real config chain (-stub-run): template_qc still resolves to 16GB, template_details_sample now resolves to 64GB. All 3 render_notebook.nf.test cases still pass. Co-Authored-By: Claude Sonnet 5 --- conf/modules.config | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/conf/modules.config b/conf/modules.config index d72982a..6582ad1 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -29,10 +29,10 @@ process { mode: params.publish_dir_mode, saveAs: { filename -> filename.equals('versions.yml') ? null : filename } ] - // template_details_compare and template_discovery_brief need far more cpu/memory than the other notebooks. - cpus = { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16 * task.attempt : 4 * task.attempt } - memory = { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 64.GB * task.attempt : 16.GB * task.attempt } - time = { ['template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16.h : 4.h } + // template_details_sample builds dense per-sample TCRdist matrices (OOM'd at 16GB); compare/discovery_brief process full VDJdb match files. + cpus = { ['template_details_sample', 'template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16 * task.attempt : 4 * task.attempt } + memory = { ['template_details_sample', 'template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 64.GB * task.attempt : 16.GB * task.attempt } + time = { ['template_details_sample', 'template_details_compare', 'template_discovery_brief'].contains(notebook.baseName) ? 16.h : 4.h } } } \ No newline at end of file