Removed trap dataset read limit

This commit is contained in:
Tom Kasper
2026-10-04 09:58:00 +01:00
parent 10b0a29d2b
commit 58974c6af1
2 changed files with 1 additions and 8 deletions
-1
View File
@@ -18,7 +18,6 @@ pod5_dataset_size : 293
pod5_stride : 12 # Use 1 in x pod5 files from the ONT NO-MISS dataset to reduce dataset size / compute requirements pod5_stride : 12 # Use 1 in x pod5 files from the ONT NO-MISS dataset to reduce dataset size / compute requirements
trap_dataset_size: 1717 trap_dataset_size: 1717
trap_pod5_stride: 72 trap_pod5_stride: 72
trap_read_limit : '-n 500000'
# custom ML # custom ML
torch_env : '../envs/torch_cpu.yaml' torch_env : '../envs/torch_cpu.yaml'
+1 -7
View File
@@ -34,11 +34,6 @@ def get_batch_range(wildcards):
if wildcards.dataset == 'trap': if wildcards.dataset == 'trap':
return [f'batch{x}' for x in range(1,config['trap_dataset_size']+1,config['trap_pod5_stride'])] return [f'batch{x}' for x in range(1,config['trap_dataset_size']+1,config['trap_pod5_stride'])]
def get_read_limit(wildcards):
if wildcards.dataset == 'trap':
return config["trap_read_limit"]
return ''
wildcard_constraints: wildcard_constraints:
batch=r"(batch)?\d+", batch=r"(batch)?\d+",
model=r'hac|fast' model=r'hac|fast'
@@ -91,10 +86,9 @@ rule basecall_pod5:
benchmarking=get_benchmarking_file, benchmarking=get_benchmarking_file,
min_qscore=get_qscore, min_qscore=get_qscore,
dorado_model=get_model_name, dorado_model=get_model_name,
n_reads=get_read_limit
shell: shell:
""" """
dorado basecaller --models-directory {config[dorado_model_dir]} {params.n_reads} --emit-fastq {params.benchmarking} {params.min_qscore} {params.dorado_model} {input.pod5} > {output} dorado basecaller --models-directory {config[dorado_model_dir]} --emit-fastq {params.benchmarking} {params.min_qscore} {params.dorado_model} {input.pod5} > {output}
""" """
rule concatenate_basecalled_fastq: rule concatenate_basecalled_fastq: