intial version
This commit is contained in:
@@ -0,0 +1,68 @@
|
||||
def get_model_name(wildcards):
|
||||
config_param=f'{wildcards.model}_model_name'
|
||||
return config[config_param]
|
||||
|
||||
def get_model_requirement(wildcards):
|
||||
return f"{config['dorado_model_dir']}/{config[f'{wildcards.model}_model_name']}
|
||||
|
||||
def get_benchmarking_file(wildcards):
|
||||
config_param = f'{wildcards.model}_benchmarks_file'
|
||||
if config[config_param]:
|
||||
return f'--batchsize-benchmarks-file {config[config_param]}'
|
||||
return ''
|
||||
|
||||
def get_qscore(wildcards):
|
||||
config_param = f'{wildcards.model}_min_q'
|
||||
if config[config_param]:
|
||||
return f'--min-qscore {config[config_param]}'
|
||||
return ''
|
||||
|
||||
def get_batch_names(input_file):
|
||||
with open(input_file,'rt') as ih:
|
||||
return [line.strip().split('/')[-1].split('.')[0] for line in ih]
|
||||
|
||||
rule download_pod5:
|
||||
output:
|
||||
'../data/raw_pod5/PBK98658_853a956f_57f83f46_{batch}.pod5'
|
||||
threads: 1
|
||||
wildcard_constraints:
|
||||
batch="\d+"
|
||||
shell:
|
||||
"""
|
||||
aws s3 cp --no-sign-request s3://ont-open-data/nomiss_96BC_P2I_SUP_2026/raw/pod5/PBK98658_853a956f_57f83f46_{wildcards.batch}.pod5 {output}
|
||||
"""
|
||||
|
||||
rule basecall_pod5:
|
||||
input:
|
||||
pod5='../data/raw_pod5/PBK98658_853a956f_57f83f46_{batch}.pod5',
|
||||
model=get_model_requirement
|
||||
output:
|
||||
'../data/basecalled_reads/{model}/PBK98658_853a956f_57f83f46_{batch}.fastq.gz'
|
||||
threads:
|
||||
32
|
||||
resources:
|
||||
gpu=1
|
||||
params:
|
||||
benchmarking=get_benchmarking_file,
|
||||
min_qscore=get_qscore,
|
||||
dorado_model=get_model_name
|
||||
wildcard_constraints:
|
||||
batch="\d+",
|
||||
model="hac|fast"
|
||||
shell:
|
||||
"""
|
||||
dorado basecaller --models-directory {config[dorado_model_dir]} --emit-fastq {params.benchmarking} {params.min_qscore} {params.dorado_model} {input.pod5} > {output}
|
||||
"""
|
||||
|
||||
rule concatenate_basecalled_fastq:
|
||||
input:
|
||||
expand('../data/basecalled_reads/{{model}}/PBK98658_853a956f_57f83f46_{batch}.fastq.gz',batch=range(1,config["pod5_dataset_size"]+1,config["pod5_stride"]))
|
||||
output:
|
||||
'../data/basecalled_reads/{model}.fastq.gz'
|
||||
threads: 1
|
||||
wildcard_constraints:
|
||||
model="hac|fast"
|
||||
shell:
|
||||
"""
|
||||
zcat {input} > {output}
|
||||
"""
|
||||
@@ -0,0 +1,84 @@
|
||||
import os
|
||||
import hashlib
|
||||
|
||||
# Prepwork
|
||||
|
||||
def get_genome_file_name(wildcards):
|
||||
return '../data/reference_genomes/'+'_'.join(x.lower() for x in wildcards.species.split())+'.fasta'
|
||||
|
||||
def get_unique_download_dir(wildcards):
|
||||
unique_id = hashlib.md5(wildcards.species.encode()).hexdigest()[:8]
|
||||
return config["tmp_dir"]+'/'+unique_id
|
||||
|
||||
def format_cli_arg(wildcards):
|
||||
# Replaces underscores with spaces for the CLI command
|
||||
return wildcards.species.replace("_", " ")
|
||||
|
||||
rule pull_ncbi_datasets_cli:
|
||||
output:
|
||||
config["datasets_binary"]
|
||||
params:
|
||||
arch=config["arch"]
|
||||
threads: 1
|
||||
shell:
|
||||
"""
|
||||
curl https://ftp.ncbi.nlm.nih.gov/pub/datasets/command-line/v2/linux-{params.arch}/datasets -o {output}
|
||||
chmod a+x {output}
|
||||
"""
|
||||
|
||||
rule pull_dorado_models:
|
||||
output:
|
||||
directory(f"{config['dorado_model_dir']}/{config['fast_model_name']}"),
|
||||
directory(f"{config['dorado_model_dir']}/{config['hac_model_name']}")
|
||||
threads: 1
|
||||
shell:
|
||||
"""
|
||||
dorado download --models-directory {config[dorado_model_dir]} --model {config[fast_model_name]}
|
||||
dorado download --models-directory {config[dorado_model_dir]} --model {config[hac_model_name]}
|
||||
"""
|
||||
|
||||
rule download_reference_genome:
|
||||
input:
|
||||
config["datasets_binary"]
|
||||
output:
|
||||
temp('../data/reference_genomes/{species}.fasta')
|
||||
params:
|
||||
api_key=os.environ['NCBI_API_KEY'],
|
||||
wd=get_unique_download_dir,
|
||||
ncbi_tax_name=lambda w: format_cli_arg(w)
|
||||
threads: 1
|
||||
shell:
|
||||
"""
|
||||
echo {wildcards.species}
|
||||
rm -rf {params.wd}
|
||||
mkdir {params.wd}
|
||||
{config[datasets_binary]} download genome taxon "{params.ncbi_tax_name}" --reference --no-progressbar --filename {params.wd}/{wildcards.species}.zip
|
||||
unzip -d {params.wd} -o {params.wd}/{wildcards.species}.zip
|
||||
mv {params.wd}/ncbi_dataset/data/*/*fna {output}
|
||||
rm -rf {params.wd}
|
||||
"""
|
||||
|
||||
rule concatenate_reference_genomes:
|
||||
input:
|
||||
expand("../data/reference_genomes/{species}.fasta",species=config["species_list"])
|
||||
output:
|
||||
"../data/reference_genomes/full_reference.fasta"
|
||||
threads: 1
|
||||
shell:
|
||||
"""
|
||||
cat {input} > {output}
|
||||
"""
|
||||
|
||||
rule create_minimap_index:
|
||||
input:
|
||||
"../data/reference_genomes/full_reference.fasta"
|
||||
output:
|
||||
"../data/reference_genomes/full_reference.mmi"
|
||||
threads:
|
||||
32
|
||||
conda:
|
||||
"../envs/minimap.yaml"
|
||||
shell:
|
||||
"""
|
||||
minimap2 -x map-ont -t {threads} -d {output} {input}
|
||||
"""
|
||||
Reference in New Issue
Block a user