Digital Trials pipeline configured for WES
Source-only snapshot of the cluster branch for WES execution. Large reference files (HPA/MANE/ensemble FASTA, model weights, ~597 MB) are omitted: they are baked into the container images at build time and mounted from the dreamdock-data PVC at runtime, and exceed the Gitea request size limit. Pipeline entry point is main.nf, which orchestrates the biotransformer, conplex and tissue modules as a single workflow. Ligand inputs are read from the eureka workspace; protein_zarr and chembl_db come from the dreamdock-data PVC.
This commit is contained in:
392
main_biotransformer.nf
Executable file
392
main_biotransformer.nf
Executable file
@@ -0,0 +1,392 @@
|
||||
nextflow.enable.dsl=2
|
||||
|
||||
|
||||
process SUPER_TRANSFORMER {
|
||||
container "${params.container_biotransformer}"
|
||||
containerOptions "${params.containerOptions}"
|
||||
publishDir "${params.outdir}/${params.project_name}/receptors", mode: 'copy'
|
||||
debug true
|
||||
|
||||
input:
|
||||
path smiles_csv
|
||||
|
||||
output:
|
||||
path "${smiles_csv.simpleName}_out.csv"
|
||||
|
||||
script:
|
||||
"""
|
||||
#!/bin/bash
|
||||
workdir=`pwd`
|
||||
cd /home/omic/biotransformer
|
||||
## Predicting Biotransformation Using the Human Super Transformer:
|
||||
# This command predicts the biotransformation of molecules from an SDF input (example.csv) using the human super transformer (superbio) and annotates the metabolites with names and database IDs (from PubChem).
|
||||
java -jar /home/omic/biotransformer/biotransformer -k pred -b superbio -isdf \$workdir/${smiles_csv.simpleName}.csv -ocsv \${workdir}/${smiles_csv.simpleName}_out.csv -osdf \${workdir}/${smiles_csv.simpleName}.sdf -a
|
||||
"""
|
||||
}
|
||||
|
||||
process HUMAN_TRANSFORMER {
|
||||
memory { params.bt_initial_memory.toFloat().GB + (task.attempt - 1) * params.bt_growth_memory.toFloat().GB }
|
||||
errorStrategy { task.attempt <= params.bt_max_retries.toInteger() ? 'retry' : params.bt_fail_action }
|
||||
maxRetries params.bt_max_retries.toInteger()
|
||||
maxForks params.bt_max_forks.toInteger() ?: null
|
||||
|
||||
container "${params.container_biotransformer}"
|
||||
// containerOptions "${params.containerOptions}"
|
||||
publishDir "${params.outdir}/${params.project_name}/1_biotransformer", mode: 'copy'
|
||||
// // Temporarily disabled debug prints
|
||||
// debug true
|
||||
// maxForks 5
|
||||
|
||||
input:
|
||||
path smiles_csv
|
||||
|
||||
output:
|
||||
path "${smiles_csv.simpleName}_out.csv"
|
||||
|
||||
script:
|
||||
"""
|
||||
#!/bin/bash
|
||||
set -e
|
||||
workdir=\$(pwd)
|
||||
cd /home/omic/biotransformer
|
||||
|
||||
# Function to get available memory percentage
|
||||
get_available_mem_percent() {
|
||||
free | awk '/Mem:/ {print int(\$7/\$2 * 100)}'
|
||||
}
|
||||
|
||||
# Function to log memory usage
|
||||
log_memory() {
|
||||
local pid=\$1
|
||||
echo "MEMLOG: Timestamp,BioTransformer Memory (MB),Available System Memory (%)" >&2
|
||||
while kill -0 \$pid 2>/dev/null; do
|
||||
local biotrans_mem=\$(ps -o rss= -p \$pid | awk '{print \$1/1024}')
|
||||
local avail_mem=\$(get_available_mem_percent)
|
||||
echo "MEMLOG: \$(date '+%Y-%m-%d %H:%M:%S'),\$biotrans_mem,\$avail_mem" >&2
|
||||
if [ \$avail_mem -lt 5 ]; then
|
||||
echo "MEMLOG: Available memory below 5%. Terminating process." >&2
|
||||
kill -15 \$pid
|
||||
wait \$pid
|
||||
echo 'PROCESS TERMINATED DUE TO LOW MEMORY' > "\${workdir}/${smiles_csv.simpleName}_out.csv"
|
||||
exit 1
|
||||
fi
|
||||
sleep 1800 # Sleep for 30 minutes
|
||||
done
|
||||
}
|
||||
|
||||
# Calculate max Java heap size (90% of available memory)
|
||||
# max_heap=\$(free -g | awk '/Mem:/ {print int(\$7 * 0.9)}')
|
||||
|
||||
#remove unnecessary data to run biotransformer
|
||||
head -n 2 "\$workdir/${smiles_csv.simpleName}.csv" > "\$workdir/${smiles_csv.simpleName}_run.csv"
|
||||
|
||||
get_container_mem_bytes() {
|
||||
if [ -f /sys/fs/cgroup/memory/memory.limit_in_bytes ]; then
|
||||
cat /sys/fs/cgroup/memory/memory.limit_in_bytes
|
||||
elif [ -f /sys/fs/cgroup/memory.max ]; then
|
||||
# cgroup v2; "max" means no limit
|
||||
val=\$(cat /sys/fs/cgroup/memory.max)
|
||||
if [ "\$val" = "max" ]; then
|
||||
# fall back to host MemTotal
|
||||
awk '/MemTotal:/ {print \$2 * 1024}' /proc/meminfo
|
||||
else
|
||||
echo "\$val"
|
||||
fi
|
||||
else
|
||||
# fallback to host MemTotal
|
||||
awk '/MemTotal:/ {print \$2 * 1024}' /proc/meminfo
|
||||
fi
|
||||
}
|
||||
|
||||
container_mem_bytes=\$(get_container_mem_bytes)
|
||||
# 95% of container memory, in MB (integer)
|
||||
heap_mb=\$(awk -v m="\$container_mem_bytes" 'BEGIN { printf "%d", (m*0.95)/(1024*1024) }')
|
||||
|
||||
echo "Using Java heap: \${heap_mb}m (95% of container limit)" >&2
|
||||
|
||||
java -Xmx"\${heap_mb}m" -jar /home/omic/biotransformer/biotransformer \
|
||||
-a -k pred -b superbio \
|
||||
-isdf "\$workdir/${smiles_csv.simpleName}_run.csv" \
|
||||
-ocsv "\${workdir}/${smiles_csv.simpleName}_out.csv" \
|
||||
-osdf "\${workdir}/${smiles_csv.simpleName}.sdf" \
|
||||
-s 100
|
||||
biotrans_pid=\$!
|
||||
|
||||
# Start memory logging in the background
|
||||
log_memory \$biotrans_pid &
|
||||
log_pid=\$!
|
||||
|
||||
# Wait for the biotransformer process to finish
|
||||
wait \$biotrans_pid
|
||||
status=\$?
|
||||
|
||||
# Stop the logging process
|
||||
kill \$log_pid 2>/dev/null || true
|
||||
|
||||
if [ ! -s "\${workdir}/${smiles_csv.simpleName}_out.csv" ]; then
|
||||
if [ \$status -ne 0 ] && [ ! -f "\${workdir}/${smiles_csv.simpleName}_out.csv" ]; then
|
||||
echo 'PROCESS FAILED' > "\${workdir}/${smiles_csv.simpleName}_out.csv"
|
||||
else
|
||||
echo 'NO METABOLITES' > "\${workdir}/${smiles_csv.simpleName}_out.csv"
|
||||
fi
|
||||
fi
|
||||
"""
|
||||
}
|
||||
|
||||
process METABOLITES_BY_MASS {
|
||||
container "${params.container_biotransformer}"
|
||||
containerOptions "${params.containerOptions}"
|
||||
publishDir "${params.outdir}/${params.project_name}/graphs", mode: 'copy'
|
||||
debug true
|
||||
|
||||
input:
|
||||
path smiles_csv
|
||||
|
||||
output:
|
||||
path "${smiles_csv.simpleName}_out.csv"
|
||||
|
||||
script:
|
||||
"""
|
||||
#!/bin/bash
|
||||
workdir=`pwd`
|
||||
cd /home/omic/biotransformer
|
||||
## Identifying Metabolites with Specific Masses:
|
||||
# This command identifies all human metabolites of compounds in example.csv with masses 292.0946 Da and 304.0946 Da (max depth = 2), with a mass tolerance of 0.01 Da. It provides annotations when available.
|
||||
java -jar /home/omic/biotransformer/biotransformer -k cid -b allHuman -isdf \$workdir/${smiles_csv.simpleName}.csv -ocsv \${workdir}/${smiles_csv.simpleName}_out.csv -osdf \${workdir}/${smiles_csv.simpleName}.sdf -s 2 -m "292.0946;304.0946" -t 0.01 -a
|
||||
"""
|
||||
}
|
||||
|
||||
process ORDERED_SEQUENCE {
|
||||
container "${params.container_biotransformer}"
|
||||
containerOptions "${params.containerOptions}"
|
||||
publishDir "${params.outdir}/${params.project_name}/screening", mode: 'copy'
|
||||
debug true
|
||||
errorStrategy 'ignore'
|
||||
|
||||
input:
|
||||
path smiles_csv
|
||||
|
||||
output:
|
||||
path "${smiles_csv.simpleName}_out.csv"
|
||||
|
||||
script:
|
||||
"""
|
||||
#!/bin/bash
|
||||
workdir=`pwd`
|
||||
cd /home/omic/biotransformer
|
||||
## Simulating an Ordered Sequence of Metabolism:
|
||||
# This command simulates an ordered sequence of metabolism for compounds in example.csv, starting with two steps of CYP450 oxidation, followed by one step of conjugation. The output is saved in an SDF file (output.sdf).
|
||||
java -jar /home/omic/biotransformer/biotransformer -isdf \$workdir/${smiles_csv.simpleName}.csv -ocsv \${workdir}/${smiles_csv.simpleName}_out.csv -osdf \${workdir}/${smiles_csv.simpleName}.sdf -k pred -q "cyp450:2; phaseII:1"
|
||||
"""
|
||||
}
|
||||
|
||||
process GET_FINAL_METABOLITES {
|
||||
container "${params.container_biotransformer}"
|
||||
containerOptions "${params.containerOptions}"
|
||||
publishDir "${params.outdir}/${params.project_name}/screening", mode: 'copy'
|
||||
// // Temporarily disabled debug prints
|
||||
// debug true
|
||||
// maxForks 1
|
||||
// errorStrategy 'ignore'
|
||||
|
||||
errorStrategy { task.attempt <= 10 ? 'retry' : 'ignore' }
|
||||
maxRetries 10
|
||||
|
||||
input:
|
||||
path smiles_csv
|
||||
|
||||
output:
|
||||
path "${smiles_csv.simpleName}.txt"
|
||||
|
||||
script:
|
||||
"""
|
||||
#!/opt/conda/envs/biotransformer/bin/python
|
||||
from rdkit import Chem
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import requests
|
||||
|
||||
#load biotransformet ouput
|
||||
biotransformer = pd.read_csv('$smiles_csv')
|
||||
|
||||
#check if metabolites exist
|
||||
if biotransformer.shape[0] == 0:
|
||||
with open('${smiles_csv.simpleName}.txt', 'w') as f:
|
||||
f.write(f"NO_METABOLITES\\n")
|
||||
|
||||
else:
|
||||
metabolite_id = biotransformer['Metabolite ID']
|
||||
#get only metabolites that are not precursors for next metabolisam step
|
||||
last_step_metabolites = biotransformer[[i not in list(biotransformer['Precursor ID']) for i in metabolite_id]]
|
||||
last_step_metabolites = last_step_metabolites.drop_duplicates()
|
||||
|
||||
#get only "active" ones
|
||||
InChIKey = last_step_metabolites['InChIKey']
|
||||
InChIKey = pd.unique(InChIKey)
|
||||
|
||||
#look if metabolite is in chembl
|
||||
chembl = []
|
||||
for i in InChIKey:
|
||||
curl_commend = f"https://www.ebi.ac.uk/chembl/api/data/molecule/{i}"
|
||||
c = requests.get(curl_commend)
|
||||
if c.status_code == 200:
|
||||
#compund exists in chembl
|
||||
chembl.append(True)
|
||||
else:
|
||||
#compund does not exists in chembl
|
||||
chembl.append(False)
|
||||
|
||||
|
||||
ChEMBL_bool = pd.DataFrame([InChIKey, chembl]).T.rename({0:'InChIKey',1:'in_ChEMBL'}, axis = 1)
|
||||
last_step_metabolites = last_step_metabolites.merge(ChEMBL_bool, how='inner', on='InChIKey')
|
||||
#if in PUBCHEM
|
||||
last_step_metabolites['in_PUBCHEM'] = ~last_step_metabolites['PUBCHEM_CID'].isna()
|
||||
##logic
|
||||
#if metabolite is not in PUBCHEM pass it to next step. We don't know if it's "active"
|
||||
#has to be in ChEMBL and PUBCHEM to have "activity"
|
||||
mask = (last_step_metabolites['in_PUBCHEM'] == False) | (np.sum(last_step_metabolites[['in_ChEMBL','in_PUBCHEM']], 1) == 2)
|
||||
|
||||
#filtered metabolites
|
||||
last_step_metabolites = last_step_metabolites[mask]
|
||||
|
||||
#check for empty smile positions
|
||||
last_step_metabolites = last_step_metabolites.reset_index()
|
||||
if last_step_metabolites['SMILES'].isnull().any():
|
||||
for i in range(len(last_step_metabolites)):
|
||||
if pd.isna(last_step_metabolites.at[i, 'SMILES']):
|
||||
last_step_metabolites.at[i, 'SMILES'] = Chem.MolToSmiles(Chem.MolFromInchi(last_step_metabolites.at[i, 'InChI']))
|
||||
|
||||
#convert biotransformer smi to canonical smi and eliminate nonsense structures
|
||||
metabolite_smi = list(np.unique(last_step_metabolites['SMILES']))
|
||||
metabolite_smi = [Chem.CanonSmiles(i) for i in metabolite_smi if Chem.MolFromSmiles(i) != None]
|
||||
|
||||
#final metabolites
|
||||
metabolite_smi_filteres = np.array(metabolite_smi)
|
||||
|
||||
#save smi
|
||||
with open('${smiles_csv.simpleName}.txt', 'w') as f:
|
||||
if len(metabolite_smi_filteres) == 0:
|
||||
f.write(f"NO_METABOLITES\\n")
|
||||
else:
|
||||
for n, line in enumerate(metabolite_smi_filteres):
|
||||
f.write(f"{line}\\tmetabol_{n}\\n")
|
||||
"""
|
||||
}
|
||||
|
||||
process GET_FINAL_METABOLITES_STATIC {
|
||||
memory 1.GB
|
||||
|
||||
container "${params.container_chembl}"
|
||||
containerOptions "${params.containerOptions}"
|
||||
publishDir "${params.outdir}/${params.project_name}/1b_final_metabolites", mode: 'copy'
|
||||
|
||||
errorStrategy { task.attempt <= 10 ? 'retry' : 'ignore' }
|
||||
maxRetries 10
|
||||
|
||||
input:
|
||||
path smiles_csv
|
||||
path chembl_db // Add ChEMBL SQLite database as input
|
||||
|
||||
output:
|
||||
path "${smiles_csv.simpleName}.txt"
|
||||
|
||||
script:
|
||||
"""
|
||||
#!/opt/conda/bin/python
|
||||
from rdkit import Chem
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import sqlite3
|
||||
|
||||
# Connect to ChEMBL database
|
||||
conn = sqlite3.connect('file:$chembl_db?mode=ro&immutable=1', uri=True)
|
||||
# conn = sqlite3.connect('$chembl_db')
|
||||
|
||||
cursor = conn.cursor()
|
||||
|
||||
# Load biotransformer output
|
||||
biotransformer = pd.read_csv('$smiles_csv')
|
||||
|
||||
# Check if metabolites exist
|
||||
if biotransformer.shape[0] == 0:
|
||||
with open('${smiles_csv.simpleName}.txt', 'w') as f:
|
||||
f.write(f"NO_METABOLITES\\n")
|
||||
|
||||
else:
|
||||
# metabolite_id = biotransformer['Metabolite ID']
|
||||
# # Get only metabolites that are not precursors for next metabolism step
|
||||
# last_step_metabolites = biotransformer[[i not in list(biotransformer['Precursor ID']) for i in metabolite_id]]
|
||||
# last_step_metabolites = last_step_metabolites.drop_duplicates()
|
||||
|
||||
# Get metabolite IDs that are NOT precursors for other metabolites
|
||||
metabolite_id = biotransformer['Metabolite ID']
|
||||
precursor_ids = set(biotransformer['Precursor ID'].dropna()) # Remove NaN values
|
||||
final_metabolite_mask = ~metabolite_id.isin(precursor_ids)
|
||||
|
||||
# Filter to get only final metabolites
|
||||
last_step_metabolites = biotransformer[final_metabolite_mask].copy()
|
||||
last_step_metabolites = last_step_metabolites.drop_duplicates()
|
||||
|
||||
# Get only "active" ones
|
||||
InChIKey = last_step_metabolites['InChIKey']
|
||||
InChIKey = pd.unique(InChIKey)
|
||||
|
||||
# Look if metabolite is in ChEMBL using local database
|
||||
chembl = []
|
||||
for inchikey in InChIKey:
|
||||
# Query ChEMBL SQLite database for the InChIKey
|
||||
# The standard InChIKey is stored in COMPOUND_STRUCTURES table
|
||||
cursor.execute(\"""
|
||||
SELECT COUNT(*)
|
||||
FROM COMPOUND_STRUCTURES
|
||||
WHERE STANDARD_INCHI_KEY = ?
|
||||
LIMIT 1
|
||||
\""", (inchikey,))
|
||||
|
||||
result = cursor.fetchone()
|
||||
chembl.append(result[0] > 0)
|
||||
|
||||
# ChEMBL_bool = pd.DataFrame([InChIKey, chembl]).T.rename({0:'InChIKey',1:'in_ChEMBL'}, axis=1)
|
||||
ChEMBL_bool = pd.DataFrame({
|
||||
'InChIKey': InChIKey,
|
||||
'in_ChEMBL': chembl
|
||||
})
|
||||
last_step_metabolites = last_step_metabolites.merge(ChEMBL_bool, how='inner', on='InChIKey')
|
||||
|
||||
# If in PUBCHEM
|
||||
last_step_metabolites['in_PUBCHEM'] = ~last_step_metabolites['PUBCHEM_CID'].isna()
|
||||
|
||||
# Logic
|
||||
# If metabolite is not in PUBCHEM pass it to next step. We don't know if it's "active"
|
||||
# Has to be in ChEMBL and PUBCHEM to have "activity"
|
||||
mask = (last_step_metabolites['in_PUBCHEM'] == False) | (np.sum(last_step_metabolites[['in_ChEMBL','in_PUBCHEM']], 1) == 2)
|
||||
|
||||
# Filtered metabolites
|
||||
last_step_metabolites = last_step_metabolites[mask]
|
||||
|
||||
# Check for empty smile positions
|
||||
last_step_metabolites = last_step_metabolites.reset_index()
|
||||
if last_step_metabolites['SMILES'].isnull().any():
|
||||
for i in range(len(last_step_metabolites)):
|
||||
if pd.isna(last_step_metabolites.at[i, 'SMILES']):
|
||||
last_step_metabolites.at[i, 'SMILES'] = Chem.MolToSmiles(Chem.MolFromInchi(last_step_metabolites.at[i, 'InChI']))
|
||||
|
||||
# Convert biotransformer smi to canonical smi and eliminate nonsense structures
|
||||
metabolite_smi = list(np.unique(last_step_metabolites['SMILES']))
|
||||
metabolite_smi = [Chem.CanonSmiles(i) for i in metabolite_smi if Chem.MolFromSmiles(i) != None]
|
||||
|
||||
# Final metabolites
|
||||
metabolite_smi_filteres = np.array(metabolite_smi)
|
||||
|
||||
# Save smi
|
||||
with open('${smiles_csv.simpleName}.txt', 'w') as f:
|
||||
if len(metabolite_smi_filteres) == 0:
|
||||
f.write(f"NO_METABOLITES\\n")
|
||||
else:
|
||||
for n, line in enumerate(metabolite_smi_filteres):
|
||||
f.write(f"{line}\\tmetabol_{n}\\n")
|
||||
|
||||
# Close database connection
|
||||
conn.close()
|
||||
"""
|
||||
}
|
||||
Reference in New Issue
Block a user