162 lines
4.9 KiB
Plaintext
162 lines
4.9 KiB
Plaintext
def get_model_name(wildcards):
|
|
config_param=f'{wildcards.model}_model_name'
|
|
return config[config_param]
|
|
|
|
def get_model_requirement(wildcards):
|
|
config_param=f'{wildcards.model}_model_name'
|
|
return config['dorado_model_dir']+'/'+config[config_param]
|
|
|
|
def get_benchmarking_file(wildcards):
|
|
config_param = f'{wildcards.model}_benchmarks_file'
|
|
if config[config_param]:
|
|
return f'--batchsize-benchmarks-file {config[config_param]}'
|
|
return ''
|
|
|
|
def get_qscore(wildcards):
|
|
config_param = f'{wildcards.model}_min_q'
|
|
if config[config_param]:
|
|
return f'--min-qscore {config[config_param]}'
|
|
return ''
|
|
|
|
def get_batch_names(input_file):
|
|
with open(input_file,'rt') as ih:
|
|
return [line.strip().split('/')[-1].split('.')[0] for line in ih]
|
|
|
|
def get_run_name(wildcards):
|
|
if wildcards.dataset == 'nomiss':
|
|
return 'PBK98658_853a956f_57f83f46'
|
|
if wildcards.dataset == 'trap':
|
|
return 'ATCC_25922_202309'
|
|
|
|
def get_batch_range(wildcards):
|
|
if wildcards.dataset == 'nomiss':
|
|
return range(1,config['pod5_dataset_size']+1,config['pod5_stride'])
|
|
if wildcards.dataset == 'trap':
|
|
return [f'batch{x}' for x in range(1,config['trap_dataset_size']+1,config['trap_pod5_stride'])]
|
|
|
|
wildcard_constraints:
|
|
batch=r"(batch)?\d+",
|
|
model=r'hac|fast'
|
|
|
|
GPU_STACK = config.get('gpu_stack','cuda')
|
|
|
|
rule download_nomiss_pod5:
|
|
output:
|
|
'../data/raw_pod5/nomiss/PBK98658_853a956f_57f83f46_{batch}.pod5'
|
|
threads: 4
|
|
|
|
shell:
|
|
"""
|
|
aws s3 cp --no-sign-request s3://ont-open-data/nomiss_96BC_P2I_SUP_2026/raw/pod5/PBK98658_853a956f_57f83f46_{wildcards.batch}.pod5 {output}
|
|
"""
|
|
|
|
rule download_trap_tar:
|
|
output:
|
|
temp('../data/raw_pod5/trap/all.tar')
|
|
threads: 4
|
|
params:
|
|
tmp='../data/raw_pod5/trap/trap_all.tar'
|
|
shell:
|
|
"""
|
|
curl -L --http1.1 -A "Mozilla/5.0" "https://api.figshare.com/v2/file/download/45408628" -o {output}
|
|
"""
|
|
rule untar_trap_folder:
|
|
input:
|
|
'../data/raw_pod5/trap/all.tar'
|
|
output:
|
|
directory('../data/raw_pod5/trap/all')
|
|
threads: 1
|
|
shell:
|
|
"""
|
|
mkdir -p {output}
|
|
tar xf {input} -C {output}
|
|
"""
|
|
|
|
rule extract_trap_pod5:
|
|
input:
|
|
'../data/raw_pod5/trap/all'
|
|
output:
|
|
'../data/raw_pod5/trap/ATCC_25922_202309_{batch}.pod5'
|
|
threads: 1
|
|
params:
|
|
input_pod5='{input[0]}/ATCC_25922__202309/ATCC_25922__202309_{batch}.pod5'
|
|
shell:
|
|
"""
|
|
mv {params.input_pod5} {output}
|
|
"""
|
|
|
|
if GPU_STACK == 'cuda':
|
|
rule basecall_pod5_dorado:
|
|
input:
|
|
pod5='../data/raw_pod5/{dataset}/{run}_{batch}.pod5',
|
|
model=get_model_requirement
|
|
output:
|
|
temp('../data/basecalled_reads/{dataset}/{model}/{run}_{batch}.fastq')
|
|
threads:
|
|
32
|
|
resources:
|
|
gpu=1
|
|
params:
|
|
benchmarking=get_benchmarking_file,
|
|
min_qscore=get_qscore,
|
|
dorado_model=get_model_name,
|
|
shell:
|
|
"""
|
|
dorado basecaller --models-directory {config[dorado_model_dir]} --emit-fastq {params.benchmarking} {params.min_qscore} {params.dorado_model} {input.pod5} > {output}
|
|
"""
|
|
elif GPU_STACK == 'rocm':
|
|
rule basecall_pod5_slorado:
|
|
input:
|
|
pod5='../data/raw_pod5/{dataset}/{run}_{batch}.pod5',
|
|
model=get_model_requirement
|
|
output:
|
|
temp('../data/basecalled_reads/{dataset}/{model}/{run}_{batch}.fastq')
|
|
threads:
|
|
32
|
|
resources:
|
|
gpu=1
|
|
shell:
|
|
"""
|
|
pod5-slorado basecaller -o {output} -t {threads} --flash=yes {input.model} {input.pod5}
|
|
"""
|
|
|
|
rule concatenate_basecalled_fastq:
|
|
input:
|
|
expand('../data/basecalled_reads/{{dataset}}/{{model}}/{run}_{batch}.fastq',batch=get_batch_range,run=get_run_name)
|
|
output:
|
|
'../data/basecalled_reads/{dataset}_{model}.fastq.gz'
|
|
threads: 1
|
|
wildcard_constraints:
|
|
model="hac|fast"
|
|
shell:
|
|
"""
|
|
cat {input} | gzip -c > {output}
|
|
"""
|
|
|
|
rule align_reads_to_reference:
|
|
input:
|
|
fastq='../data/basecalled_reads/{dataset}_{model}.fastq.gz',
|
|
ref='../data/reference_genomes/full_reference.mmi'
|
|
output:
|
|
'../data/aligned_reads/{dataset}_{model}_to_genome.sam'
|
|
threads: 32
|
|
conda:
|
|
'../envs/minimap.yaml'
|
|
shell:
|
|
"""
|
|
minimap2 -ax map-ont -t {threads} {input.ref} {input.fastq} > {output}
|
|
"""
|
|
|
|
rule convert_sam_to_bam:
|
|
input:
|
|
'../data/aligned_reads/{dataset}_{model}_to_genome.sam'
|
|
output:
|
|
'../data/aligned_reads/{dataset}_{model}_to_genome.sorted.bam'
|
|
threads: 32
|
|
conda:
|
|
'../envs/samtools.yaml'
|
|
shell:
|
|
"""
|
|
samtools view -b -@ {threads} {input} | samtools sort -@ {threads} > {output}
|
|
samtools index -@ {threads} {output}
|
|
""" |