更新 .gitignore,添加 06.gene_trees 目录;重构 macse.sh 脚本,移除 fs_lr 参数;删除多个不再使用的脚本;添加 check_frameshift.py 和 get_og_seqs.py 脚本以处理阅读框移位和提取单拷贝OG序列;更新 pixi.toml 和 pixi.lock 文件以添加 paml 依赖。
This commit is contained in:
@@ -0,0 +1,128 @@
|
||||
#! /usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
"""
|
||||
Check and process frameshift in alignment files.
|
||||
"""
|
||||
|
||||
import shutil
|
||||
import pandas as pd
|
||||
from pathlib import Path
|
||||
from Bio import SeqIO
|
||||
from Bio.SeqRecord import SeqRecord
|
||||
from Bio.Seq import Seq
|
||||
import sys
|
||||
import argparse
|
||||
|
||||
|
||||
def parse_frameshift_list(fs_list: Path) -> pd.DataFrame:
|
||||
"""
|
||||
Parse a frameshift list file into a pandas DataFrame.
|
||||
|
||||
Args:
|
||||
fs_list (Path): Path to the frameshift list file.
|
||||
|
||||
Returns:
|
||||
pd.DataFrame: DataFrame containing the frameshift information.
|
||||
"""
|
||||
try:
|
||||
df = pd.read_csv(fs_list, sep=",", header=0)
|
||||
except Exception as e:
|
||||
print(f"Error reading frameshift list file {fs_list}: {e}")
|
||||
sys.exit(1)
|
||||
return df
|
||||
|
||||
|
||||
def get_alignment_files(aln_dir: Path, ext: str) -> list[Path]:
|
||||
"""
|
||||
Get a list of alignment files in a directory with a specific extension.
|
||||
|
||||
Args:
|
||||
aln_dir (Path): Path to the directory containing alignment files.
|
||||
ext (str): Extension of the alignment files.
|
||||
|
||||
Returns:
|
||||
list[Path]: List of Paths to the alignment files.
|
||||
"""
|
||||
try:
|
||||
aln_files = list(aln_dir.glob(f"*{ext}"))
|
||||
except Exception as e:
|
||||
print(f"Error accessing alignment files in {aln_dir}: {e}")
|
||||
sys.exit(1)
|
||||
return aln_files
|
||||
|
||||
|
||||
def main(alignment_dir: str, ext: str, frameshift_list: str, outdir: str):
|
||||
aln_dir = Path(alignment_dir)
|
||||
fs_list_path = Path(frameshift_list)
|
||||
out_dir = Path(outdir)
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
fs_df = parse_frameshift_list(fs_list_path)
|
||||
aln_files = get_alignment_files(aln_dir, ext)
|
||||
good_aln_count = 0
|
||||
keep_fs_count = 0
|
||||
discard_count = 0
|
||||
|
||||
for file in aln_files:
|
||||
if str(file) not in fs_df["alignment_file"].values.tolist():
|
||||
try:
|
||||
shutil.copy(file, out_dir / file.name)
|
||||
good_aln_count += 1
|
||||
continue
|
||||
except Exception as e:
|
||||
print(f"Error copying file {file} to {out_dir}: {e}")
|
||||
sys.exit(1)
|
||||
if (
|
||||
fs_df.loc[
|
||||
fs_df["alignment_file"] == str(file), "possible_attributions"
|
||||
].values[0]
|
||||
== "Framshift only in Ziziphus jujuba or Elaeagnus pungens"
|
||||
):
|
||||
try:
|
||||
records = []
|
||||
for record in SeqIO.parse(file, "fasta"):
|
||||
seq = str(record.seq).replace("!", "-")
|
||||
id = record.id
|
||||
records.append(SeqRecord(seq=Seq(seq), id=id, description=""))
|
||||
print(
|
||||
f"Keep the alignment file with frameshift: {file}. Due to frameshift only in outgroup."
|
||||
)
|
||||
keep_fs_count += 1
|
||||
SeqIO.write(records, out_dir / file.name, "fasta")
|
||||
except Exception as e:
|
||||
print(f"Error processing file {file}: {e}")
|
||||
sys.exit(1)
|
||||
else:
|
||||
discard_count += 1
|
||||
print("Frameshift processing completed successfully.")
|
||||
print(f"Number of good alignments copied: {good_aln_count}")
|
||||
print(f"Number of alignments with frameshift kept: {keep_fs_count}")
|
||||
print(f"Number of alignments with frameshift discarded: {discard_count}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Check and process frameshift in alignment files."
|
||||
)
|
||||
parser.add_argument(
|
||||
"-a",
|
||||
"--alignment_dir",
|
||||
required=True,
|
||||
help="Directory containing alignment files",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-e", "--ext", default=".nal", help="Extension of alignment files"
|
||||
)
|
||||
parser.add_argument(
|
||||
"-f",
|
||||
"--frameshift_list",
|
||||
required=True,
|
||||
help="CSV file listing frameshift information",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-o", "--outdir", required=True, help="Output directory for processed files"
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
main(args.alignment_dir, args.ext, args.frameshift_list, args.outdir)
|
||||
@@ -0,0 +1,148 @@
|
||||
#! /usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
"""
|
||||
Extract Single Copy OG sequences from a comprehensive FASTA file based on OG definitions from results of orthofinder.
|
||||
"""
|
||||
|
||||
import sys
|
||||
from Bio import SeqIO
|
||||
from Bio.SeqRecord import SeqRecord
|
||||
from Bio.Seq import Seq
|
||||
from pathlib import Path
|
||||
import argparse
|
||||
|
||||
|
||||
def get_og_fastas(og_dir: Path, ext: str) -> dict[str, list[str]]:
|
||||
"""
|
||||
Get a dictionary of OGs and their corresponding sequence IDs from FASTA files in a directory.
|
||||
|
||||
Args:
|
||||
og_dir (Path): Path to the directory containing OG FASTA files.
|
||||
ext (str): Extension of the FASTA files. Default is ".fa".
|
||||
|
||||
Returns:
|
||||
dict: Dictionary with OG names as keys and lists of sequence IDs as values.
|
||||
"""
|
||||
og_dict: dict[str, list[str]] = {}
|
||||
try:
|
||||
for fasta_file in og_dir.glob(f"*{ext}"):
|
||||
og_name = fasta_file.stem
|
||||
seq_ids: list[str] = []
|
||||
for record in SeqIO.parse(fasta_file, "fasta"):
|
||||
seq_ids.append(record.id)
|
||||
og_dict[og_name] = seq_ids
|
||||
except Exception as e:
|
||||
print(f"Error processing OG FASTA files in {og_dir}: {e}")
|
||||
sys.exit(1)
|
||||
return og_dict
|
||||
|
||||
|
||||
def parse_all_fasta(fasta_file: Path) -> dict[str, Seq]:
|
||||
"""
|
||||
Parse a FASTA file and return a dictionary of sequence IDs and their corresponding Seq objects.
|
||||
|
||||
Args:
|
||||
fasta_file (Path): Path to the FASTA file.
|
||||
Returns:
|
||||
dict: Dictionary with sequence IDs as keys and Seq objects as values.
|
||||
"""
|
||||
seq_dict: dict[str, Seq] = {}
|
||||
try:
|
||||
for record in SeqIO.parse(fasta_file, "fasta"):
|
||||
seq_dict[record.id] = record.seq
|
||||
except Exception as e:
|
||||
print(f"Error parsing FASTA file {fasta_file}: {e}")
|
||||
sys.exit(1)
|
||||
return seq_dict
|
||||
|
||||
|
||||
def output_og_seqs(
|
||||
all_seq_dict: dict[str, Seq], og_dict: dict[str, list[str]], output_dir: Path
|
||||
):
|
||||
"""
|
||||
Output sequences for each OG into separate FASTA files, remove gene id and only keep taxon name.
|
||||
|
||||
Args:
|
||||
all_seq_dict (dict): Dictionary of all sequences with sequence IDs as keys.
|
||||
og_dict (dict): Dictionary of OGs with OG names as keys and lists of sequence IDs as values.
|
||||
output_dir (Path): Path to the output directory.
|
||||
"""
|
||||
for og_name, seq_ids in og_dict.items():
|
||||
og_seqs: list[SeqRecord] = []
|
||||
for seq_id in seq_ids:
|
||||
if seq_id in all_seq_dict:
|
||||
seq_record = SeqRecord(
|
||||
all_seq_dict[seq_id], id=seq_id.split("@")[0], description=""
|
||||
)
|
||||
og_seqs.append(seq_record)
|
||||
else:
|
||||
print(f"Warning: Sequence ID {seq_id} not found in all sequences.")
|
||||
output_fasta = output_dir / f"{og_name}.fa"
|
||||
try:
|
||||
SeqIO.write(og_seqs, output_fasta, "fasta")
|
||||
except Exception as e:
|
||||
print(f"Error writing to FASTA file {output_fasta}: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def write_og_list(og_dict: dict[str, list[str]]):
|
||||
"""
|
||||
Write the OG list to a text file.
|
||||
|
||||
Args:
|
||||
og_dict (dict): Dictionary of OGs with OG names as keys and lists of sequence IDs as values.
|
||||
"""
|
||||
try:
|
||||
with open("og_list.tsv", "w") as f:
|
||||
for og_name, seq_ids in og_dict.items():
|
||||
line = f"{og_name}\t" + "\t".join(seq_ids) + "\n"
|
||||
f.write(line)
|
||||
print("OG list written to og_list.tsv")
|
||||
except Exception as e:
|
||||
print(f"Error writing OG list to file: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def main(og_dir_path: str, all_fasta_path: str, output_dir_path: str, ext: str = ".fa"):
|
||||
og_dir = Path(og_dir_path)
|
||||
all_fasta = Path(all_fasta_path)
|
||||
output_dir = Path(output_dir_path)
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
og_dict = get_og_fastas(og_dir, ext)
|
||||
all_seqs = parse_all_fasta(all_fasta)
|
||||
output_og_seqs(all_seqs, og_dict, output_dir)
|
||||
write_og_list(og_dict)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Extract OG sequences from a comprehensive FASTA file based on OG definitions."
|
||||
)
|
||||
parser.add_argument(
|
||||
"-d",
|
||||
"--og_dir",
|
||||
required=True,
|
||||
help="Directory containing OG FASTA files.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-a",
|
||||
"--all_fasta",
|
||||
required=True,
|
||||
help="FASTA file containing all sequences.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-o",
|
||||
"--output_dir",
|
||||
required=True,
|
||||
help="Output directory for OG FASTA files.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-e",
|
||||
"--ext",
|
||||
default=".fa",
|
||||
help="Extension of OG FASTA files (default: .fa).",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
main(args.og_dir, args.all_fasta, args.output_dir, args.ext)
|
||||
Executable
+74
@@ -0,0 +1,74 @@
|
||||
#! /usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
"""
|
||||
Get primary CDS sequences from a FASTA file containing multiple CDS per gene.
|
||||
"""
|
||||
|
||||
from Bio import SeqIO
|
||||
from Bio.SeqRecord import SeqRecord
|
||||
import re
|
||||
import argparse
|
||||
import sys
|
||||
|
||||
|
||||
def get_primary_cds(input_fasta, output_fasta):
|
||||
primary_cds_records = []
|
||||
gene = ""
|
||||
length = 0
|
||||
seq = None
|
||||
id = None
|
||||
try:
|
||||
for record in SeqIO.parse(input_fasta, "fasta"):
|
||||
seq_len = len(record.seq)
|
||||
desc = record.description
|
||||
match = re.search(r"\[gene=(\S+)\]", desc)
|
||||
if match:
|
||||
gene_name = match.group(1)
|
||||
else:
|
||||
# Skip if gene name not found
|
||||
continue
|
||||
|
||||
if gene_name != gene:
|
||||
# new gene encountered
|
||||
# print(f"Processing gene: {gene_name}")
|
||||
if length > 0:
|
||||
# this is not the first record, save the previous longest record
|
||||
primary_cds_record = SeqRecord(
|
||||
seq, id=id, description=f"[gene={gene}]"
|
||||
)
|
||||
primary_cds_records.append(primary_cds_record)
|
||||
gene = gene_name
|
||||
seq = record.seq
|
||||
id = record.id
|
||||
length = seq_len
|
||||
else:
|
||||
# same gene, check length
|
||||
if seq_len > length:
|
||||
seq = record.seq
|
||||
id = record.id
|
||||
length = seq_len
|
||||
# after loop, save the last gene
|
||||
if gene and length > 0:
|
||||
primary_cds_record = SeqRecord(seq, id=id, description=f"[gene={gene}]")
|
||||
primary_cds_records.append(primary_cds_record)
|
||||
SeqIO.write(primary_cds_records, output_fasta, "fasta")
|
||||
print(f"Primary CDS sequences written to {args.output_fasta}")
|
||||
except Exception as e:
|
||||
print(f"Error processing FASTA file {input_fasta}: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Extract primary CDS sequences from a FASTA file."
|
||||
)
|
||||
parser.add_argument(
|
||||
"-i", "--input_fasta", help="Input FASTA file containing CDS sequences."
|
||||
)
|
||||
parser.add_argument(
|
||||
"-o", "--output_fasta", help="Output FASTA file to write primary CDS sequences."
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
get_primary_cds(args.input_fasta, args.output_fasta)
|
||||
@@ -1,7 +1,5 @@
|
||||
#! /bin/bash
|
||||
|
||||
FS_LR=7
|
||||
|
||||
if [ "$#" -ne 3 ]; then
|
||||
echo "Usage: $0 <reliable_seq> <less_reliable_seq> <stem>"
|
||||
echo "Perform MACSE alignment on given sequences"
|
||||
@@ -16,7 +14,7 @@ stem=$3
|
||||
echo "Command:"
|
||||
echo "macse -prog alignSequences -seq $seq -seq_lr ${seq_lr}"
|
||||
echo " -out_NT ${stem}.nal -out_AA ${stem}.pal"
|
||||
echo " -optim 2 -max_refine_iter 3 -local_realign_init 0.2 -fs_lr $FS_LR"
|
||||
echo " -optim 2 -max_refine_iter 3 -local_realign_init 0.2"
|
||||
} >alignSequences.log
|
||||
macse -prog alignSequences \
|
||||
-seq "$seq" -seq_lr "${seq_lr}" \
|
||||
@@ -24,5 +22,4 @@ macse -prog alignSequences \
|
||||
-optim 2 \
|
||||
-max_refine_iter 3 \
|
||||
-local_realign_init 0.2 \
|
||||
-fs_lr $FS_LR \
|
||||
>>alignSequences.log 2>&1
|
||||
|
||||
Reference in New Issue
Block a user