added limit for number of trap reads to process

This commit is contained in:
Tom Kasper
2026-10-03 15:35:38 +01:00
parent 88406e584e
commit 0fdb5fa07b
2 changed files with 12 additions and 5 deletions
+1
View File
@@ -16,6 +16,7 @@ fast_min_q: 8 # empty string or 0 to disable
pod5_dataset: "nomiss_96BC_P2I_SUP_2026" pod5_dataset: "nomiss_96BC_P2I_SUP_2026"
pod5_dataset_size : 293 pod5_dataset_size : 293
pod5_stride : 12 # Use 1 in x pod5 files from the ONT NO-MISS dataset to reduce dataset size / compute requirements pod5_stride : 12 # Use 1 in x pod5 files from the ONT NO-MISS dataset to reduce dataset size / compute requirements
trap_read_limit : '-n 500000'
# custom ML # custom ML
torch_env : '../envs/torch_cpu.yaml' torch_env : '../envs/torch_cpu.yaml'
+11 -5
View File
@@ -34,12 +34,17 @@ def get_batch_range(wildcards):
if wildcards.dataset == 'trap': if wildcards.dataset == 'trap':
return [0] return [0]
def get_read_limit(wildcards):
if wildcards.dataset == 'trap':
return config["trap_read_limit"]
return ''
rule download_nomiss_pod5: rule download_nomiss_pod5:
output: output:
'../data/raw_pod5/nomiss/PBK98658_853a956f_57f83f46_{batch}.pod5' '../data/raw_pod5/nomiss/PBK98658_853a956f_57f83f46_{batch}.pod5'
threads: 4 threads: 4
wildcard_constraints: wildcard_constraints:
batch="\d+" batch=r"\d+"
shell: shell:
""" """
aws s3 cp --no-sign-request s3://ont-open-data/nomiss_96BC_P2I_SUP_2026/raw/pod5/PBK98658_853a956f_57f83f46_{wildcards.batch}.pod5 {output} aws s3 cp --no-sign-request s3://ont-open-data/nomiss_96BC_P2I_SUP_2026/raw/pod5/PBK98658_853a956f_57f83f46_{wildcards.batch}.pod5 {output}
@@ -50,7 +55,7 @@ rule download_trap_pod5:
'../data/raw_pod5/trap/ATCC_25922_202309_0.pod5' '../data/raw_pod5/trap/ATCC_25922_202309_0.pod5'
threads: 4 threads: 4
wildcard_constraints: wildcard_constraints:
batch='\d+' batch=r'\d+'
shell: shell:
""" """
curl -L "https://api.figshare.com/v2/file/download/45408628" -o {output} curl -L "https://api.figshare.com/v2/file/download/45408628" -o {output}
@@ -68,13 +73,14 @@ rule basecall_pod5:
params: params:
benchmarking=get_benchmarking_file, benchmarking=get_benchmarking_file,
min_qscore=get_qscore, min_qscore=get_qscore,
dorado_model=get_model_name dorado_model=get_model_name,
n_reads=get_read_limit
wildcard_constraints: wildcard_constraints:
batch="\d+", batch=r"\d+",
model="hac|fast" model="hac|fast"
shell: shell:
""" """
dorado basecaller --models-directory {config[dorado_model_dir]} --emit-fastq {params.benchmarking} {params.min_qscore} {params.dorado_model} {input.pod5} > {output} dorado basecaller --models-directory {config[dorado_model_dir]} {params.n_reads} --emit-fastq {params.benchmarking} {params.min_qscore} {params.dorado_model} {input.pod5} > {output}
""" """
rule concatenate_basecalled_fastq: rule concatenate_basecalled_fastq: