Files
digital-trial/main_biotransformer.nf
Olamide Isreal 724c7d8987 Mount the dreamdock-data PVC in the k8s profile, and probe for it
The PVC holding chembl_36.db and protein_seq.zarr was never mounted into task
pods: WES's injected profile.config supplies no storage config (verified across
a full run's logs). Without /mnt/dreamdock-data, sqlite3 raises 'unable to open
database file' and every GET_FINAL_METABOLITES_STATIC task exits 1.

Restore storageClaimName/storageMountPath, and add a probe that reports what is
visible before opening sqlite so the failure names its own cause.
2026-07-28 11:04:46 +01:00

421 lines
17 KiB
Plaintext
Executable File

nextflow.enable.dsl=2
process SUPER_TRANSFORMER {
container "${params.container_biotransformer}"
containerOptions "${params.containerOptions}"
publishDir "${params.outdir}/${params.project_name}/receptors", mode: 'copy'
debug true
input:
path smiles_csv
output:
path "${smiles_csv.simpleName}_out.csv"
script:
"""
#!/bin/bash
workdir=`pwd`
cd /home/omic/biotransformer
## Predicting Biotransformation Using the Human Super Transformer:
# This command predicts the biotransformation of molecules from an SDF input (example.csv) using the human super transformer (superbio) and annotates the metabolites with names and database IDs (from PubChem).
java -jar /home/omic/biotransformer/biotransformer -k pred -b superbio -isdf \$workdir/${smiles_csv.simpleName}.csv -ocsv \${workdir}/${smiles_csv.simpleName}_out.csv -osdf \${workdir}/${smiles_csv.simpleName}.sdf -a
"""
}
process HUMAN_TRANSFORMER {
memory { params.bt_initial_memory.toFloat().GB + (task.attempt - 1) * params.bt_growth_memory.toFloat().GB }
errorStrategy { task.attempt <= params.bt_max_retries.toInteger() ? 'retry' : params.bt_fail_action }
maxRetries params.bt_max_retries.toInteger()
maxForks params.bt_max_forks.toInteger() ?: null
container "${params.container_biotransformer}"
// containerOptions "${params.containerOptions}"
publishDir "${params.outdir}/${params.project_name}/1_biotransformer", mode: 'copy'
// // Temporarily disabled debug prints
// debug true
// maxForks 5
input:
path smiles_csv
output:
path "${smiles_csv.simpleName}_out.csv"
script:
"""
#!/bin/bash
set -e
workdir=\$(pwd)
cd /home/omic/biotransformer
# Function to get available memory percentage
get_available_mem_percent() {
free | awk '/Mem:/ {print int(\$7/\$2 * 100)}'
}
# Function to log memory usage
log_memory() {
local pid=\$1
echo "MEMLOG: Timestamp,BioTransformer Memory (MB),Available System Memory (%)" >&2
while kill -0 \$pid 2>/dev/null; do
local biotrans_mem=\$(ps -o rss= -p \$pid | awk '{print \$1/1024}')
local avail_mem=\$(get_available_mem_percent)
echo "MEMLOG: \$(date '+%Y-%m-%d %H:%M:%S'),\$biotrans_mem,\$avail_mem" >&2
if [ \$avail_mem -lt 5 ]; then
echo "MEMLOG: Available memory below 5%. Terminating process." >&2
kill -15 \$pid
wait \$pid
echo 'PROCESS TERMINATED DUE TO LOW MEMORY' > "\${workdir}/${smiles_csv.simpleName}_out.csv"
exit 1
fi
sleep 1800 # Sleep for 30 minutes
done
}
# Calculate max Java heap size (90% of available memory)
# max_heap=\$(free -g | awk '/Mem:/ {print int(\$7 * 0.9)}')
#remove unnecessary data to run biotransformer
head -n 2 "\$workdir/${smiles_csv.simpleName}.csv" > "\$workdir/${smiles_csv.simpleName}_run.csv"
get_container_mem_bytes() {
if [ -f /sys/fs/cgroup/memory/memory.limit_in_bytes ]; then
cat /sys/fs/cgroup/memory/memory.limit_in_bytes
elif [ -f /sys/fs/cgroup/memory.max ]; then
# cgroup v2; "max" means no limit
val=\$(cat /sys/fs/cgroup/memory.max)
if [ "\$val" = "max" ]; then
# fall back to host MemTotal
awk '/MemTotal:/ {print \$2 * 1024}' /proc/meminfo
else
echo "\$val"
fi
else
# fallback to host MemTotal
awk '/MemTotal:/ {print \$2 * 1024}' /proc/meminfo
fi
}
container_mem_bytes=\$(get_container_mem_bytes)
# 95% of container memory, in MB (integer)
heap_mb=\$(awk -v m="\$container_mem_bytes" 'BEGIN { printf "%d", (m*0.95)/(1024*1024) }')
echo "Using Java heap: \${heap_mb}m (95% of container limit)" >&2
java -Xmx"\${heap_mb}m" -jar /home/omic/biotransformer/biotransformer \
-a -k pred -b superbio \
-isdf "\$workdir/${smiles_csv.simpleName}_run.csv" \
-ocsv "\${workdir}/${smiles_csv.simpleName}_out.csv" \
-osdf "\${workdir}/${smiles_csv.simpleName}.sdf" \
-s 100
biotrans_pid=\$!
# Start memory logging in the background
log_memory \$biotrans_pid &
log_pid=\$!
# Wait for the biotransformer process to finish
wait \$biotrans_pid
status=\$?
# Stop the logging process
kill \$log_pid 2>/dev/null || true
if [ ! -s "\${workdir}/${smiles_csv.simpleName}_out.csv" ]; then
if [ \$status -ne 0 ] && [ ! -f "\${workdir}/${smiles_csv.simpleName}_out.csv" ]; then
echo 'PROCESS FAILED' > "\${workdir}/${smiles_csv.simpleName}_out.csv"
else
echo 'NO METABOLITES' > "\${workdir}/${smiles_csv.simpleName}_out.csv"
fi
fi
"""
}
process METABOLITES_BY_MASS {
container "${params.container_biotransformer}"
containerOptions "${params.containerOptions}"
publishDir "${params.outdir}/${params.project_name}/graphs", mode: 'copy'
debug true
input:
path smiles_csv
output:
path "${smiles_csv.simpleName}_out.csv"
script:
"""
#!/bin/bash
workdir=`pwd`
cd /home/omic/biotransformer
## Identifying Metabolites with Specific Masses:
# This command identifies all human metabolites of compounds in example.csv with masses 292.0946 Da and 304.0946 Da (max depth = 2), with a mass tolerance of 0.01 Da. It provides annotations when available.
java -jar /home/omic/biotransformer/biotransformer -k cid -b allHuman -isdf \$workdir/${smiles_csv.simpleName}.csv -ocsv \${workdir}/${smiles_csv.simpleName}_out.csv -osdf \${workdir}/${smiles_csv.simpleName}.sdf -s 2 -m "292.0946;304.0946" -t 0.01 -a
"""
}
process ORDERED_SEQUENCE {
container "${params.container_biotransformer}"
containerOptions "${params.containerOptions}"
publishDir "${params.outdir}/${params.project_name}/screening", mode: 'copy'
debug true
errorStrategy 'ignore'
input:
path smiles_csv
output:
path "${smiles_csv.simpleName}_out.csv"
script:
"""
#!/bin/bash
workdir=`pwd`
cd /home/omic/biotransformer
## Simulating an Ordered Sequence of Metabolism:
# This command simulates an ordered sequence of metabolism for compounds in example.csv, starting with two steps of CYP450 oxidation, followed by one step of conjugation. The output is saved in an SDF file (output.sdf).
java -jar /home/omic/biotransformer/biotransformer -isdf \$workdir/${smiles_csv.simpleName}.csv -ocsv \${workdir}/${smiles_csv.simpleName}_out.csv -osdf \${workdir}/${smiles_csv.simpleName}.sdf -k pred -q "cyp450:2; phaseII:1"
"""
}
process GET_FINAL_METABOLITES {
container "${params.container_biotransformer}"
containerOptions "${params.containerOptions}"
publishDir "${params.outdir}/${params.project_name}/screening", mode: 'copy'
// // Temporarily disabled debug prints
// debug true
// maxForks 1
// errorStrategy 'ignore'
errorStrategy { task.attempt <= 10 ? 'retry' : 'ignore' }
maxRetries 10
input:
path smiles_csv
output:
path "${smiles_csv.simpleName}.txt"
script:
"""
#!/opt/conda/envs/biotransformer/bin/python
from rdkit import Chem
import numpy as np
import pandas as pd
import requests
#load biotransformet ouput
biotransformer = pd.read_csv('$smiles_csv')
#check if metabolites exist
if biotransformer.shape[0] == 0:
with open('${smiles_csv.simpleName}.txt', 'w') as f:
f.write(f"NO_METABOLITES\\n")
else:
metabolite_id = biotransformer['Metabolite ID']
#get only metabolites that are not precursors for next metabolisam step
last_step_metabolites = biotransformer[[i not in list(biotransformer['Precursor ID']) for i in metabolite_id]]
last_step_metabolites = last_step_metabolites.drop_duplicates()
#get only "active" ones
InChIKey = last_step_metabolites['InChIKey']
InChIKey = pd.unique(InChIKey)
#look if metabolite is in chembl
chembl = []
for i in InChIKey:
curl_commend = f"https://www.ebi.ac.uk/chembl/api/data/molecule/{i}"
c = requests.get(curl_commend)
if c.status_code == 200:
#compund exists in chembl
chembl.append(True)
else:
#compund does not exists in chembl
chembl.append(False)
ChEMBL_bool = pd.DataFrame([InChIKey, chembl]).T.rename({0:'InChIKey',1:'in_ChEMBL'}, axis = 1)
last_step_metabolites = last_step_metabolites.merge(ChEMBL_bool, how='inner', on='InChIKey')
#if in PUBCHEM
last_step_metabolites['in_PUBCHEM'] = ~last_step_metabolites['PUBCHEM_CID'].isna()
##logic
#if metabolite is not in PUBCHEM pass it to next step. We don't know if it's "active"
#has to be in ChEMBL and PUBCHEM to have "activity"
mask = (last_step_metabolites['in_PUBCHEM'] == False) | (np.sum(last_step_metabolites[['in_ChEMBL','in_PUBCHEM']], 1) == 2)
#filtered metabolites
last_step_metabolites = last_step_metabolites[mask]
#check for empty smile positions
last_step_metabolites = last_step_metabolites.reset_index()
if last_step_metabolites['SMILES'].isnull().any():
for i in range(len(last_step_metabolites)):
if pd.isna(last_step_metabolites.at[i, 'SMILES']):
last_step_metabolites.at[i, 'SMILES'] = Chem.MolToSmiles(Chem.MolFromInchi(last_step_metabolites.at[i, 'InChI']))
#convert biotransformer smi to canonical smi and eliminate nonsense structures
metabolite_smi = list(np.unique(last_step_metabolites['SMILES']))
metabolite_smi = [Chem.CanonSmiles(i) for i in metabolite_smi if Chem.MolFromSmiles(i) != None]
#final metabolites
metabolite_smi_filteres = np.array(metabolite_smi)
#save smi
with open('${smiles_csv.simpleName}.txt', 'w') as f:
if len(metabolite_smi_filteres) == 0:
f.write(f"NO_METABOLITES\\n")
else:
for n, line in enumerate(metabolite_smi_filteres):
f.write(f"{line}\\tmetabol_{n}\\n")
"""
}
process GET_FINAL_METABOLITES_STATIC {
// The ChEMBL DB is ~9.2 GB. It is passed as a `val` path (not a staged `path`
// input) so Nextflow does not copy it into every task work directory — with
// concurrent tasks that multiplied into hundreds of GB of I/O. The DB lives on
// the dreamdock-data PVC and is opened read-only/immutable, so sharing one copy
// across tasks is safe.
memory { params.chembl_initial_memory.toInteger().GB + (task.attempt - 1) * params.chembl_growth_memory.toInteger().GB }
container "${params.container_chembl}"
containerOptions "${params.containerOptions}"
publishDir "${params.outdir}/${params.project_name}/1b_final_metabolites", mode: 'copy'
errorStrategy { task.attempt <= params.chembl_max_retries.toInteger() ? 'retry' : params.chembl_fail_action }
maxRetries { params.chembl_max_retries.toInteger() }
input:
path smiles_csv
val chembl_db // Path to the ChEMBL SQLite DB on the PVC (not staged into the workdir)
output:
path "${smiles_csv.simpleName}.txt"
script:
"""
#!/opt/conda/bin/python
from rdkit import Chem
import numpy as np
import pandas as pd
import sqlite3
import os, sys
# --- diagnostic probe -------------------------------------------------
# The ChEMBL DB lives on the dreamdock-data PVC. If that PVC is not mounted
# into the task pod, sqlite3 fails with a generic "unable to open database
# file" that is indistinguishable from a corrupt/locked DB. Report what is
# actually visible before touching sqlite, so the failure names its cause.
_db = '$chembl_db'
print('[probe] chembl_db path : %s' % _db, file=sys.stderr)
print('[probe] exists : %s' % os.path.exists(_db), file=sys.stderr)
for _p in ('/mnt/dreamdock-data', os.path.dirname(_db)):
print('[probe] dir %-22s exists=%s' % (_p, os.path.isdir(_p)), file=sys.stderr)
if os.path.exists(_db):
print('[probe] size : %d bytes' % os.path.getsize(_db), file=sys.stderr)
print('[probe] readable : %s' % os.access(_db, os.R_OK), file=sys.stderr)
else:
sys.exit(
"[probe] FATAL: %s is not visible inside this task container.\n"
"[probe] The dreamdock-data PVC is not mounted. Check that the k8s\n"
"[probe] profile sets k8s.storageClaimName/storageMountPath and that\n"
"[probe] the WES-injected profile.config does not override them."
% _db
)
# --- end probe --------------------------------------------------------
# Connect to ChEMBL database
conn = sqlite3.connect('file:$chembl_db?mode=ro&immutable=1', uri=True)
# conn = sqlite3.connect('$chembl_db')
cursor = conn.cursor()
# Load biotransformer output
biotransformer = pd.read_csv('$smiles_csv')
# Check if metabolites exist
if biotransformer.shape[0] == 0:
with open('${smiles_csv.simpleName}.txt', 'w') as f:
f.write(f"NO_METABOLITES\\n")
else:
# metabolite_id = biotransformer['Metabolite ID']
# # Get only metabolites that are not precursors for next metabolism step
# last_step_metabolites = biotransformer[[i not in list(biotransformer['Precursor ID']) for i in metabolite_id]]
# last_step_metabolites = last_step_metabolites.drop_duplicates()
# Get metabolite IDs that are NOT precursors for other metabolites
metabolite_id = biotransformer['Metabolite ID']
precursor_ids = set(biotransformer['Precursor ID'].dropna()) # Remove NaN values
final_metabolite_mask = ~metabolite_id.isin(precursor_ids)
# Filter to get only final metabolites
last_step_metabolites = biotransformer[final_metabolite_mask].copy()
last_step_metabolites = last_step_metabolites.drop_duplicates()
# Get only "active" ones
InChIKey = last_step_metabolites['InChIKey']
InChIKey = pd.unique(InChIKey)
# Look if metabolite is in ChEMBL using local database
chembl = []
for inchikey in InChIKey:
# Query ChEMBL SQLite database for the InChIKey
# The standard InChIKey is stored in COMPOUND_STRUCTURES table
cursor.execute(\"""
SELECT COUNT(*)
FROM COMPOUND_STRUCTURES
WHERE STANDARD_INCHI_KEY = ?
LIMIT 1
\""", (inchikey,))
result = cursor.fetchone()
chembl.append(result[0] > 0)
# ChEMBL_bool = pd.DataFrame([InChIKey, chembl]).T.rename({0:'InChIKey',1:'in_ChEMBL'}, axis=1)
ChEMBL_bool = pd.DataFrame({
'InChIKey': InChIKey,
'in_ChEMBL': chembl
})
last_step_metabolites = last_step_metabolites.merge(ChEMBL_bool, how='inner', on='InChIKey')
# If in PUBCHEM
last_step_metabolites['in_PUBCHEM'] = ~last_step_metabolites['PUBCHEM_CID'].isna()
# Logic
# If metabolite is not in PUBCHEM pass it to next step. We don't know if it's "active"
# Has to be in ChEMBL and PUBCHEM to have "activity"
mask = (last_step_metabolites['in_PUBCHEM'] == False) | (np.sum(last_step_metabolites[['in_ChEMBL','in_PUBCHEM']], 1) == 2)
# Filtered metabolites
last_step_metabolites = last_step_metabolites[mask]
# Check for empty smile positions
last_step_metabolites = last_step_metabolites.reset_index()
if last_step_metabolites['SMILES'].isnull().any():
for i in range(len(last_step_metabolites)):
if pd.isna(last_step_metabolites.at[i, 'SMILES']):
last_step_metabolites.at[i, 'SMILES'] = Chem.MolToSmiles(Chem.MolFromInchi(last_step_metabolites.at[i, 'InChI']))
# Convert biotransformer smi to canonical smi and eliminate nonsense structures
metabolite_smi = list(np.unique(last_step_metabolites['SMILES']))
metabolite_smi = [Chem.CanonSmiles(i) for i in metabolite_smi if Chem.MolFromSmiles(i) != None]
# Final metabolites
metabolite_smi_filteres = np.array(metabolite_smi)
# Save smi
with open('${smiles_csv.simpleName}.txt', 'w') as f:
if len(metabolite_smi_filteres) == 0:
f.write(f"NO_METABOLITES\\n")
else:
for n, line in enumerate(metabolite_smi_filteres):
f.write(f"{line}\\tmetabol_{n}\\n")
# Close database connection
conn.close()
"""
}