This commit is contained in:
2025-11-25 00:28:51 +08:00
commit eb3f16c30e
406 changed files with 91653 additions and 0 deletions
+144
View File
@@ -0,0 +1,144 @@
#! /usr/bin/env python3
#!/usr/bin/env python3
"""
CDS to Protein Converter with Internal Stop Codon Filtering
This script processes CDS sequences from a FASTA file, translates them to protein sequences,
checks for internal stop codons, and outputs clean CDS and protein sequences.
"""
import sys
from Bio import SeqIO
from Bio.SeqRecord import SeqRecord
def translate_cds_and_filter(
input_fasta, output_clean_cds, output_clean_protein, translation_table=1
):
"""
Main processing function: Translates CDS sequences and filters those with internal stop codons[2](@ref)
Parameters:
input_fasta: Path to input CDS sequences FASTA file
output_clean_cds: Path for output clean CDS sequences
output_clean_protein: Path for output protein sequences
translation_table: Genetic code table number (default: 1 = Standard)
Returns:
Tuple of (clean_cds_count, removed_count)
"""
clean_cds_records = [] # Store CDS sequences without internal stop codons
clean_protein_records = [] # Store corresponding protein sequences
removed_count = 0 # Count of removed sequences
total_count = 0 # Total sequences processed
print(f"Processing file: {input_fasta}")
# Process each sequence in the input FASTA file
for record in SeqIO.parse(input_fasta, "fasta"):
total_count += 1
cds_seq = record.seq
seq_id = record.id
# Check if sequence length is multiple of 3
if len(cds_seq) % 3 != 0:
print(f"Warning: Sequence {seq_id} length is not multiple of 3, skipping.")
removed_count += 1
continue
try:
# Translate CDS to protein sequence (including stop codon '*')
protein_seq = cds_seq.translate(table=translation_table, to_stop=False)
protein_str = str(protein_seq)
# Find all stop codon positions in the protein sequence
stop_positions = [i for i, aa in enumerate(protein_str) if aa == "*"]
has_internal_stop = False
# Check if any stop codon is not at the end (internal stop)
if stop_positions:
last_position = len(protein_str) - 1
# Internal stop exists if stop codon is found not at the very end
if any(pos != last_position for pos in stop_positions):
has_internal_stop = True
if has_internal_stop:
# Skip sequences with internal stop codons
print(
f"Warning: Removing sequence {seq_id}: Internal stop codon detected"
)
removed_count += 1
else:
# Create clean protein sequence (remove terminal stop codon if present)
if protein_str.endswith("*"):
protein_seq_clean = protein_seq[:-1] # Remove terminal stop codon
else:
protein_seq_clean = protein_seq
# Create protein sequence record
protein_record = SeqRecord(
seq=protein_seq_clean, id=seq_id, description=record.description
)
# Add to results
clean_cds_records.append(record)
clean_protein_records.append(protein_record)
except Exception as e:
print(f"Error processing sequence {seq_id}: {e}")
removed_count += 1
continue
# Write output files if we have valid sequences
if clean_cds_records:
SeqIO.write(clean_cds_records, output_clean_cds, "fasta")
SeqIO.write(clean_protein_records, output_clean_protein, "fasta")
print("\nProcessing completed successfully!")
print(f"Total input sequences: {total_count}")
print(f"Sequences retained: {len(clean_cds_records)}")
print(f"Sequences removed: {removed_count}")
print(f"Clean CDS sequences saved to: {output_clean_cds}")
print(f"Protein sequences saved to: {output_clean_protein}")
return len(clean_cds_records), removed_count
else:
print("Warning: No sequences passed filtering. Please check input file format.")
return 0, removed_count
def main():
"""Main command-line interface function"""
if len(sys.argv) != 3:
print("Usage: python cds_to_protein_filter.py input.fasta output_stem")
print("Arguments:")
print(" input.fasta Input CDS sequences FASTA file")
print(" output_stem Stem for output files ")
sys.exit(1)
input_file = sys.argv[1]
output_stem = sys.argv[2]
output_cds_file = f"{output_stem}.cds.fa"
output_protein_file = f"{output_stem}.pep.fa"
# Verify input file exists
try:
with open(input_file, "r"):
pass
except FileNotFoundError:
print(f"Error: Input file {input_file} not found!")
sys.exit(1)
except IOError as e:
print(f"Error reading input file {input_file}: {e}")
sys.exit(1)
# Execute processing
try:
translate_cds_and_filter(input_file, output_cds_file, output_protein_file)
except Exception as e:
print(f"Fatal error during processing: {e}")
sys.exit(1)
if __name__ == "__main__":
main()
+215
View File
@@ -0,0 +1,215 @@
#! /usr/bin/env python3
import os
import re
import argparse
from Bio import SeqIO
from pathlib import Path
def parse_fasta(fasta_file_path):
"""
Parse a FASTA file and return a list of sequence ids.
Args:
fasta_file_path: Path to the FASTA file
Returns:
list: List of sequence ids
"""
sequence_ids = []
try:
for record in SeqIO.parse(fasta_file_path, "fasta"):
sequence_ids.append(record.id)
except Exception as e:
print(f"Error parsing FASTA file {fasta_file_path}: {e}")
return sequence_ids
def parse_hmmer_tbl(tbl_file_path):
"""
Parse HMMER tbl format result file and extract best hit information
Args:
tbl_file_path: Path to the tbl file
Returns:
dict: Best hit information
"""
best_hit = {}
try:
with open(tbl_file_path, "r") as f:
for line in f:
# Skip comment lines
if line.startswith("#"):
continue
# Split line (tbl format is typically space or tab separated)
parts = re.split(r"\s+", line.strip())
if len(parts) < 5:
continue
# Extract key information: target name, E-value, score, etc.
# tbl format columns: target name, target accession, query name, query accession, E-value, score, etc.
target_name = parts[0]
e_value = float(parts[4]) # Full sequence E-value
# If it's a new sequence or we found a better hit (lower E-value)
if not best_hit.keys() or e_value < best_hit["e_value"]:
best_hit = {
"e_value": e_value,
"target_name": target_name,
}
except Exception as e:
print(f"Error parsing tbl file {tbl_file_path}: {e}")
return {}
return best_hit
def find_corresponding_files(fasta_dir, tbl_dir):
"""
Find corresponding FASTA and tbl file pairs
Args:
fasta_dir: Directory containing FASTA files
tbl_dir: Directory containing tbl result files
Returns:
list: List of (fasta_file_path, tbl_file_path) tuples
"""
file_pairs = []
# Get all FASTA files
fasta_files = {}
for fasta_path in Path(fasta_dir).glob("*.fa"):
stem = fasta_path.stem
fasta_files[stem] = fasta_path
# Find corresponding tbl files
for stem, fasta_path in fasta_files.items():
tbl_name = f"{stem}.fa.hmmsearch.tblout"
tbl_path = Path(tbl_dir) / tbl_name
if tbl_path.exists():
file_pairs.append((fasta_path, tbl_path))
else:
print(f"Warning: No corresponding tbl file found for {stem}")
return file_pairs
def add_best_hit_to_og_list(file_pair):
"""
Add best hit information to FASTA file
Args:
file_pair: Tuple of (fasta_path, tbl_path)
Returns:
tuple: (stem, seq_ids)
"""
fasta_path, tbl_path = file_pair
# Parse tbl file to get best hits
best_hit = parse_hmmer_tbl(tbl_path)
if not best_hit:
print(f"Warning: No valid hits found in {tbl_path}")
return fasta_path.stem, []
best_seq = best_hit["target_name"]
seq_ids = parse_fasta(fasta_path)
seq_ids.append(best_seq)
return fasta_path.stem, seq_ids
def process_all_files(fasta_dir, tbl_dir, output_file):
"""
Process all FASTA and tbl file pairs
Args:
fasta_dir: Directory containing FASTA files
tbl_dir: Directory containing tbl result files
output_file: Output file path
"""
print("Starting file processing...")
print(f"FASTA directory: {fasta_dir}")
print(f"tbl directory: {tbl_dir}")
print(f"Output file: {output_file}")
print("-" * 50)
og_list = {}
# Find corresponding file pairs
file_pairs = find_corresponding_files(fasta_dir, tbl_dir)
if not file_pairs:
print("No corresponding FASTA-tbl file pairs found")
return
print(f"Found {len(file_pairs)} file pairs to process")
# Process each file pair
for i, file_pair in enumerate(file_pairs, 1):
print(f"\nProcessing pair {i}/{len(file_pairs)}:")
stem, seq_ids = add_best_hit_to_og_list(file_pair)
if not seq_ids:
print(f"Skipping {stem} due to no valid hits")
continue
og_list[stem] = seq_ids
with open(output_file, "w") as out_f:
for og, ids in og_list.items():
out_f.write(f"{og}\t" + "\t".join(ids) + "\n")
print(f"\nProcessing completed! All results saved to: {output_file}")
def main(fasta_directory, tbl_directory, output_file):
"""
Main function - take paths and run processing
Args:
fasta_directory: Folder containing FASTA files
tbl_directory: Folder containing tbl result files
output_file: Output file path
"""
# Check if input directories exist
if not os.path.exists(fasta_directory):
print(f"Error: FASTA directory does not exist {fasta_directory}")
return
if not os.path.exists(tbl_directory):
print(f"Error: tbl directory does not exist {tbl_directory}")
return
# Process all files
process_all_files(fasta_directory, tbl_directory, output_file)
if __name__ == "__main__":
# Parse command-line arguments and call main
parser = argparse.ArgumentParser(
description="Add HMMER best hit into orthologs FASTA."
)
parser.add_argument(
"-f",
"--fasta_directory",
required=True,
help="Directory containing FASTA files",
)
parser.add_argument(
"-t",
"--tbl_directory",
required=True,
help="Directory containing tbl result files",
)
parser.add_argument(
"-o",
"--output_file",
required=True,
help="File to write output seqname",
)
args = parser.parse_args()
main(args.fasta_directory, args.tbl_directory, args.output_file)
+92
View File
@@ -0,0 +1,92 @@
#! /usr/bin/env python3
import os
import argparse
from Bio import SeqIO
from Bio.SeqRecord import SeqRecord
from pathlib import Path
def parse_fasta(fasta_file_path):
"""
Parse a FASTA file and return a list of SeqRecord.
Args:
fasta_file_path: Path to the FASTA file
Returns:
list: List of SeqRecord
"""
try:
records = list(SeqIO.parse(fasta_file_path, "fasta"))
return records
except Exception as e:
print(f"Error parsing FASTA file {fasta_file_path}: {e}")
return []
def ogs_to_fasta(ogs_name, seq_list, source_records, output_dir):
"""
Convert OGS list to FASTA format.
Args:
ogs_name: Name of the OGS
seq_list: List of sequence IDs
source_records: List of SeqRecord from source FASTA
output_dir: Directory to save the output FASTA file
"""
output_path = Path(output_dir) / f"{ogs_name}.fa"
seq_dict = {record.id: record for record in source_records}
with open(output_path, "w") as out_f:
for seq_id in seq_list:
if seq_id in seq_dict:
updated_id = seq_id.split("@")[0]
updated_record = SeqRecord(
seq_dict[seq_id].seq,
id=updated_id,
description="",
)
SeqIO.write(updated_record, out_f, "fasta")
else:
print(f"Warning: Sequence ID {seq_id} not found in source records.")
def process_ogs_list(ogs_file, source_fasta, output_dir):
"""
Process OGS list file and convert each OGS to FASTA format.
Args:
ogs_file: Path to the OGS list file
source_fasta: Path to the source FASTA file
output_dir: Directory to save the output FASTA files
"""
source_records = parse_fasta(source_fasta)
with open(ogs_file, "r") as f:
for line in f:
parts = line.strip().split("\t")
if len(parts) < 2:
continue
ogs_name = parts.pop(0)
ogs_to_fasta(ogs_name, parts, source_records, output_dir)
print(f"Processed OGS: {ogs_name}")
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Convert OGS list to FASTA format.")
parser.add_argument(
"-i", "--input_ogs", required=True, help="Path to the OGS list file"
)
parser.add_argument(
"-s", "--source_fasta", required=True, help="Path to the source FASTA file"
)
parser.add_argument(
"-o",
"--output_dir",
required=True,
help="Directory to save the output FASTA files",
)
args = parser.parse_args()
os.makedirs(args.output_dir, exist_ok=True)
process_ogs_list(args.input_ogs, args.source_fasta, args.output_dir)
+226
View File
@@ -0,0 +1,226 @@
#!/usr/bin/env python
from xvfbwrapper import Xvfb
import argparse
import re
import json
from ete3 import Tree, TreeStyle, NodeStyle, faces
helptext= '''
Generate the "Pie Chart" representation of gene tree conflict from Smith et al. 2015 from
the output of phyparts, the bipartition summary software described in the same paper.
The input files include three files produced by PhyParts, and a file containing a species
tree in Newick format (likely, the tree used for PhyParts). The output is an SVG containing
the phylogeny along with pie charts at each node.
Requirements:
Python 3
ete3
'''
vdisplay = Xvfb()
vdisplay.start()
#Read in species tree and convert to ultrametric
#Match phyparts nodes to ete3 nodes
def get_phyparts_nodes(sptree_fn,phyparts_root):
sptree = Tree(sptree_fn)
sptree.convert_to_ultrametric()
phyparts_node_key = [line for line in open(phyparts_root+".node.key")]
subtrees_dict = {n.split()[0]:Tree(n.split()[1]+";") for n in phyparts_node_key}
subtrees_topids = {}
for x in subtrees_dict:
subtrees_topids[x] = subtrees_dict[x].get_topology_id()
#print(subtrees_topids['1'])
#print()
for node in sptree.traverse():
node_topid = node.get_topology_id()
if "Takakia_4343a" in node.get_leaf_names():
print(node_topid)
print(node)
for subtree in subtrees_dict:
if node_topid == subtrees_topids[subtree]:
node.name = subtree
return sptree,subtrees_dict,subtrees_topids
#Summarize concordance and conflict from Phyparts
def get_concord_and_conflict(phyparts_root,subtrees_dict,subtrees_topids):
with open(phyparts_root + ".concon.tre") as phyparts_trees:
concon_tree = Tree(phyparts_trees.readline())
conflict_tree = Tree(phyparts_trees.readline())
concord_dict = {}
conflict_dict = {}
for node in concon_tree.traverse():
node_topid = node.get_topology_id()
for subtree in subtrees_dict:
if node_topid == subtrees_topids[subtree]:
concord_dict[subtree] = node.support
for node in conflict_tree.traverse():
node_topid = node.get_topology_id()
for subtree in subtrees_dict:
if node_topid == subtrees_topids[subtree]:
conflict_dict[subtree] = node.support
return concord_dict, conflict_dict
#Generate Pie Chart data
def get_pie_chart_data(phyparts_root,total_genes,concord_dict,conflict_dict):
phyparts_hist = [line for line in open(phyparts_root + ".hist")]
phyparts_pies = {}
phyparts_dict = {}
for n in phyparts_hist:
n = n.split(",")
tot_genes = float(n.pop(-1))
node_name = n.pop(0)[4:]
concord = float(n.pop(0))
concord = concord_dict[node_name]
all_conflict = conflict_dict[node_name]
if len(n) > 0:
most_conflict = max([float(x) for x in n])
else:
most_conflict = 0.0
adj_concord = (concord/total_genes) * 100
adj_most_conflict = (most_conflict/total_genes) * 100
other_conflict = (all_conflict - most_conflict) / total_genes * 100
the_rest = (total_genes - concord - all_conflict) / total_genes * 100
pie_list = [adj_concord,adj_most_conflict,other_conflict,the_rest]
phyparts_pies[node_name] = pie_list
phyparts_dict[node_name] = [int(round(concord,0)),int(round(tot_genes-concord,0))]
return phyparts_dict, phyparts_pies
def node_text_layout(mynode):
F = faces.TextFace(mynode.name,fsize=20)
faces.add_face_to_node(F,mynode,0,position="branch-right")
#convert internal phypartspiechart.py data files to csv and export to current directory (for use as ggtree tree data in R)
def pie_data_to_csv(phyparts_dict, phyparts_pies):
phyparts_dist_bin = {}
phyparts_pies_bin = {}
dist_replaced = {}
pies_replaced = {}
phyparts_dist_bin = json.dumps(phyparts_dist)
phyparts_pies_bin = json.dumps(phyparts_pies)
dist_replaced = re.sub(r'{',r'node,concord,genes-concord\n',phyparts_dist_bin)
dist_replaced = re.sub(r'"(\d*)":\s\[(\d*),\s(\d*)\],\s', r'\1,\2,\3\n', dist_replaced)
dist_replaced = re.sub(r'"(\d*)":\s\[(\d*),\s(\d*)\]}', r'\1,\2,\3', dist_replaced)
pies_replaced = re.sub(r'{',r'node,adj_concord,adj_most_conflict,other_conflict,the_rest\n',phyparts_pies_bin)
pies_replaced = re.sub(r'"(\d*)":\s\[(\d*.\d*),\s(\d*.\d*),\s(\d*.\d*),\s(\d*.\d*)\],\s', r'\1,\2,\3,\4,\5\n', pies_replaced)
pies_replaced = re.sub(r'"(\d*)":\s\[(\d*.\d*),\s(\d*.\d*),\s(\d*.\d*),\s(\d*.\d*)\]}', r'\1,\2,\3,\4,\5', pies_replaced)
with open('phyparts_dist.csv','w') as file:
for line in dist_replaced:
file.write(line)
with open('phyparts_pies.csv','w') as file:
for line in pies_replaced:
file.write(line)
parser = argparse.ArgumentParser(description=helptext,formatter_class=argparse.RawTextHelpFormatter)
parser.add_argument('species_tree',help="Newick formatted species tree topology.")
parser.add_argument('phyparts_root',help="File root name used for Phyparts.")
parser.add_argument('num_genes',type=int,default=0,help="Number of total gene trees. Used to properly scale pie charts.")
parser.add_argument('--taxon_subst',help="Comma-delimted file to translate tip names.")
parser.add_argument("--svg_name",help="File name for SVG generated by script",default="pies.svg")
parser.add_argument("--show_nodes",help="Also show tree with nodes labeled same as PhyParts",action="store_true",default=False)
parser.add_argument("--colors",help="Four colors of the pie chart: concordance (blue) top conflict (green), other conflict (red), no signal (gray)",nargs="+",default=["blue","green","red","dark gray"])
parser.add_argument("--no_ladderize",help="Do not ladderize the input species tree.",action="store_true",default=False)
parser.add_argument("--to_csv",help="Output data files to csv for import into ggtree in R",action="store_true",default=False)
args = parser.parse_args()
if args.no_ladderize:
ladderize=False
else:
ladderize=True
plot_tree,subtrees_dict,subtrees_topids = get_phyparts_nodes(args.species_tree, args.phyparts_root)
#print(subtrees_dict)
concord_dict, conflict_dict = get_concord_and_conflict(args.phyparts_root,subtrees_dict,subtrees_topids)
phyparts_dist, phyparts_pies = get_pie_chart_data(args.phyparts_root,args.num_genes,concord_dict,conflict_dict)
if args.taxon_subst:
taxon_subst = {line.split(",")[0]:line.rstrip().split(",")[1] for line in open(args.taxon_subst,'U')}
for leaf in plot_tree.get_leaves():
try:
leaf.name = taxon_subst[leaf.name]
except KeyError:
print(leaf.name)
continue
def phyparts_pie_layout(mynode):
if mynode.name in phyparts_pies:
pie= faces.PieChartFace(phyparts_pies[mynode.name],
#colors=COLOR_SCHEMES["set1"],
colors = args.colors,
width=50, height=50)
pie.border.width = None
pie.opacity = 1
faces.add_face_to_node(pie,mynode, 0, position="branch-right")
concord_text = faces.TextFace(str(int(concord_dict[mynode.name]))+' ',fsize=20)
conflict_text = faces.TextFace(str(int(conflict_dict[mynode.name]))+' ',fsize=20)
faces.add_face_to_node(concord_text,mynode,0,position = "branch-top")
faces.add_face_to_node(conflict_text,mynode,0,position="branch-bottom")
else:
F = faces.TextFace(mynode.name,fsize=20)
faces.add_face_to_node(F,mynode,0,position="aligned")
#Plot Pie Chart
ts = TreeStyle()
ts.show_leaf_name = False
ts.layout_fn = phyparts_pie_layout
nstyle = NodeStyle()
nstyle["size"] = 0
for n in plot_tree.traverse():
n.set_style(nstyle)
n.img_style["vt_line_width"] = 0
ts.draw_guiding_lines = True
ts.guiding_lines_color = "black"
ts.guiding_lines_type = 0
ts.scale = 30
ts.branch_vertical_margin = 10
plot_tree.convert_to_ultrametric()
if args.to_csv:
pie_data_to_csv(phyparts_dist, phyparts_pies)
if ladderize:
plot_tree.ladderize(direction=1)
my_svg = plot_tree.render(args.svg_name,tree_style=ts,w=595,dpi=300)
if args.show_nodes:
node_style = TreeStyle()
node_style.show_leaf_name=False
node_style.layout_fn = node_text_layout
plot_tree.render("tree_nodes.pdf",tree_style=node_style)
vdisplay.stop()
@@ -0,0 +1,2 @@
#!/usr/bin/env python3
"samtools consensus -r NC_058887.1:11845-11988 --show-del yes -A -H 0.3 -m simple --show-ins yes -f fasta EP.sorted.bam"
+71
View File
@@ -0,0 +1,71 @@
#!/usr/bin/env python3
"""
Trinity FASTA Sequence Renaming Script
Function: Rename sequences in FASTA file to format: [prefix@sequence_number]
"""
import sys
import os
def rename_fasta_sequences(input_file, prefix, output_file=None):
"""
Rename sequence headers in a FASTA file
Parameters:
input_file: Input FASTA filename
prefix: Prefix for sequence names
output_file: Output filename (optional, defaults to input_file_renamed.fasta)
"""
# Set output filename
if output_file is None:
file_base, file_ext = os.path.splitext(input_file)
output_file = f"{file_base}_renamed{file_ext}"
# Counter for sequences
seq_count = 0
try:
with open(input_file, 'r') as fin, open(output_file, 'w') as fout:
for line in fin:
if line.startswith('>'):
# Sequence header line: rename it
seq_count += 1
new_name = f">{prefix}@mrna_{seq_count}\n"
fout.write(new_name)
else:
# Sequence data line: write as-is
fout.write(line)
print(f"Successfully renamed {seq_count} sequences")
print(f"Input file: {input_file}")
print(f"Output file: {output_file}")
print(f"Naming format: {prefix}@mrna_number")
except FileNotFoundError:
print(f"Error: Input file '{input_file}' not found")
sys.exit(1)
except Exception as e:
print(f"Error processing file: {e}")
sys.exit(1)
def main():
"""Main function"""
if len(sys.argv) < 3:
print("Usage: python script.py <fasta_file> <prefix> [output_file]")
print("Example: python script.py sequences.fasta Gene new_sequences.fasta")
sys.exit(1)
input_file = sys.argv[1]
prefix = sys.argv[2]
output_file = sys.argv[3] if len(sys.argv) > 3 else None
# Verify input file exists
if not os.path.isfile(input_file):
print(f"Error: File '{input_file}' does not exist")
sys.exit(1)
rename_fasta_sequences(input_file, prefix, output_file)
if __name__ == "__main__":
main()