20251125
This commit is contained in:
Binary file not shown.
+144
@@ -0,0 +1,144 @@
|
||||
#! /usr/bin/env python3
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
CDS to Protein Converter with Internal Stop Codon Filtering
|
||||
|
||||
This script processes CDS sequences from a FASTA file, translates them to protein sequences,
|
||||
checks for internal stop codons, and outputs clean CDS and protein sequences.
|
||||
"""
|
||||
|
||||
import sys
|
||||
from Bio import SeqIO
|
||||
from Bio.SeqRecord import SeqRecord
|
||||
|
||||
|
||||
def translate_cds_and_filter(
|
||||
input_fasta, output_clean_cds, output_clean_protein, translation_table=1
|
||||
):
|
||||
"""
|
||||
Main processing function: Translates CDS sequences and filters those with internal stop codons[2](@ref)
|
||||
|
||||
Parameters:
|
||||
input_fasta: Path to input CDS sequences FASTA file
|
||||
output_clean_cds: Path for output clean CDS sequences
|
||||
output_clean_protein: Path for output protein sequences
|
||||
translation_table: Genetic code table number (default: 1 = Standard)
|
||||
|
||||
Returns:
|
||||
Tuple of (clean_cds_count, removed_count)
|
||||
"""
|
||||
clean_cds_records = [] # Store CDS sequences without internal stop codons
|
||||
clean_protein_records = [] # Store corresponding protein sequences
|
||||
removed_count = 0 # Count of removed sequences
|
||||
total_count = 0 # Total sequences processed
|
||||
|
||||
print(f"Processing file: {input_fasta}")
|
||||
|
||||
# Process each sequence in the input FASTA file
|
||||
for record in SeqIO.parse(input_fasta, "fasta"):
|
||||
total_count += 1
|
||||
cds_seq = record.seq
|
||||
seq_id = record.id
|
||||
|
||||
# Check if sequence length is multiple of 3
|
||||
if len(cds_seq) % 3 != 0:
|
||||
print(f"Warning: Sequence {seq_id} length is not multiple of 3, skipping.")
|
||||
removed_count += 1
|
||||
continue
|
||||
|
||||
try:
|
||||
# Translate CDS to protein sequence (including stop codon '*')
|
||||
protein_seq = cds_seq.translate(table=translation_table, to_stop=False)
|
||||
protein_str = str(protein_seq)
|
||||
|
||||
# Find all stop codon positions in the protein sequence
|
||||
stop_positions = [i for i, aa in enumerate(protein_str) if aa == "*"]
|
||||
has_internal_stop = False
|
||||
|
||||
# Check if any stop codon is not at the end (internal stop)
|
||||
if stop_positions:
|
||||
last_position = len(protein_str) - 1
|
||||
# Internal stop exists if stop codon is found not at the very end
|
||||
if any(pos != last_position for pos in stop_positions):
|
||||
has_internal_stop = True
|
||||
|
||||
if has_internal_stop:
|
||||
# Skip sequences with internal stop codons
|
||||
print(
|
||||
f"Warning: Removing sequence {seq_id}: Internal stop codon detected"
|
||||
)
|
||||
removed_count += 1
|
||||
else:
|
||||
# Create clean protein sequence (remove terminal stop codon if present)
|
||||
if protein_str.endswith("*"):
|
||||
protein_seq_clean = protein_seq[:-1] # Remove terminal stop codon
|
||||
else:
|
||||
protein_seq_clean = protein_seq
|
||||
|
||||
# Create protein sequence record
|
||||
protein_record = SeqRecord(
|
||||
seq=protein_seq_clean, id=seq_id, description=record.description
|
||||
)
|
||||
|
||||
# Add to results
|
||||
clean_cds_records.append(record)
|
||||
clean_protein_records.append(protein_record)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error processing sequence {seq_id}: {e}")
|
||||
removed_count += 1
|
||||
continue
|
||||
|
||||
# Write output files if we have valid sequences
|
||||
if clean_cds_records:
|
||||
SeqIO.write(clean_cds_records, output_clean_cds, "fasta")
|
||||
SeqIO.write(clean_protein_records, output_clean_protein, "fasta")
|
||||
|
||||
print("\nProcessing completed successfully!")
|
||||
print(f"Total input sequences: {total_count}")
|
||||
print(f"Sequences retained: {len(clean_cds_records)}")
|
||||
print(f"Sequences removed: {removed_count}")
|
||||
print(f"Clean CDS sequences saved to: {output_clean_cds}")
|
||||
print(f"Protein sequences saved to: {output_clean_protein}")
|
||||
|
||||
return len(clean_cds_records), removed_count
|
||||
else:
|
||||
print("Warning: No sequences passed filtering. Please check input file format.")
|
||||
return 0, removed_count
|
||||
|
||||
|
||||
def main():
|
||||
"""Main command-line interface function"""
|
||||
if len(sys.argv) != 3:
|
||||
print("Usage: python cds_to_protein_filter.py input.fasta output_stem")
|
||||
print("Arguments:")
|
||||
print(" input.fasta Input CDS sequences FASTA file")
|
||||
print(" output_stem Stem for output files ")
|
||||
sys.exit(1)
|
||||
|
||||
input_file = sys.argv[1]
|
||||
output_stem = sys.argv[2]
|
||||
output_cds_file = f"{output_stem}.cds.fa"
|
||||
output_protein_file = f"{output_stem}.pep.fa"
|
||||
|
||||
# Verify input file exists
|
||||
try:
|
||||
with open(input_file, "r"):
|
||||
pass
|
||||
except FileNotFoundError:
|
||||
print(f"Error: Input file {input_file} not found!")
|
||||
sys.exit(1)
|
||||
except IOError as e:
|
||||
print(f"Error reading input file {input_file}: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
# Execute processing
|
||||
try:
|
||||
translate_cds_and_filter(input_file, output_cds_file, output_protein_file)
|
||||
except Exception as e:
|
||||
print(f"Fatal error during processing: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+215
@@ -0,0 +1,215 @@
|
||||
#! /usr/bin/env python3
|
||||
import os
|
||||
import re
|
||||
import argparse
|
||||
from Bio import SeqIO
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def parse_fasta(fasta_file_path):
|
||||
"""
|
||||
Parse a FASTA file and return a list of sequence ids.
|
||||
Args:
|
||||
fasta_file_path: Path to the FASTA file
|
||||
Returns:
|
||||
list: List of sequence ids
|
||||
"""
|
||||
sequence_ids = []
|
||||
try:
|
||||
for record in SeqIO.parse(fasta_file_path, "fasta"):
|
||||
sequence_ids.append(record.id)
|
||||
except Exception as e:
|
||||
print(f"Error parsing FASTA file {fasta_file_path}: {e}")
|
||||
return sequence_ids
|
||||
|
||||
|
||||
def parse_hmmer_tbl(tbl_file_path):
|
||||
"""
|
||||
Parse HMMER tbl format result file and extract best hit information
|
||||
|
||||
Args:
|
||||
tbl_file_path: Path to the tbl file
|
||||
|
||||
Returns:
|
||||
dict: Best hit information
|
||||
"""
|
||||
best_hit = {}
|
||||
|
||||
try:
|
||||
with open(tbl_file_path, "r") as f:
|
||||
for line in f:
|
||||
# Skip comment lines
|
||||
if line.startswith("#"):
|
||||
continue
|
||||
|
||||
# Split line (tbl format is typically space or tab separated)
|
||||
parts = re.split(r"\s+", line.strip())
|
||||
if len(parts) < 5:
|
||||
continue
|
||||
|
||||
# Extract key information: target name, E-value, score, etc.
|
||||
# tbl format columns: target name, target accession, query name, query accession, E-value, score, etc.
|
||||
target_name = parts[0]
|
||||
e_value = float(parts[4]) # Full sequence E-value
|
||||
|
||||
# If it's a new sequence or we found a better hit (lower E-value)
|
||||
if not best_hit.keys() or e_value < best_hit["e_value"]:
|
||||
best_hit = {
|
||||
"e_value": e_value,
|
||||
"target_name": target_name,
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error parsing tbl file {tbl_file_path}: {e}")
|
||||
return {}
|
||||
|
||||
return best_hit
|
||||
|
||||
|
||||
def find_corresponding_files(fasta_dir, tbl_dir):
|
||||
"""
|
||||
Find corresponding FASTA and tbl file pairs
|
||||
|
||||
Args:
|
||||
fasta_dir: Directory containing FASTA files
|
||||
tbl_dir: Directory containing tbl result files
|
||||
|
||||
Returns:
|
||||
list: List of (fasta_file_path, tbl_file_path) tuples
|
||||
"""
|
||||
file_pairs = []
|
||||
|
||||
# Get all FASTA files
|
||||
fasta_files = {}
|
||||
|
||||
for fasta_path in Path(fasta_dir).glob("*.fa"):
|
||||
stem = fasta_path.stem
|
||||
fasta_files[stem] = fasta_path
|
||||
|
||||
# Find corresponding tbl files
|
||||
for stem, fasta_path in fasta_files.items():
|
||||
tbl_name = f"{stem}.fa.hmmsearch.tblout"
|
||||
tbl_path = Path(tbl_dir) / tbl_name
|
||||
if tbl_path.exists():
|
||||
file_pairs.append((fasta_path, tbl_path))
|
||||
else:
|
||||
print(f"Warning: No corresponding tbl file found for {stem}")
|
||||
|
||||
return file_pairs
|
||||
|
||||
|
||||
def add_best_hit_to_og_list(file_pair):
|
||||
"""
|
||||
Add best hit information to FASTA file
|
||||
|
||||
Args:
|
||||
file_pair: Tuple of (fasta_path, tbl_path)
|
||||
Returns:
|
||||
tuple: (stem, seq_ids)
|
||||
"""
|
||||
fasta_path, tbl_path = file_pair
|
||||
|
||||
# Parse tbl file to get best hits
|
||||
best_hit = parse_hmmer_tbl(tbl_path)
|
||||
|
||||
if not best_hit:
|
||||
print(f"Warning: No valid hits found in {tbl_path}")
|
||||
return fasta_path.stem, []
|
||||
|
||||
best_seq = best_hit["target_name"]
|
||||
seq_ids = parse_fasta(fasta_path)
|
||||
seq_ids.append(best_seq)
|
||||
|
||||
return fasta_path.stem, seq_ids
|
||||
|
||||
|
||||
def process_all_files(fasta_dir, tbl_dir, output_file):
|
||||
"""
|
||||
Process all FASTA and tbl file pairs
|
||||
|
||||
Args:
|
||||
fasta_dir: Directory containing FASTA files
|
||||
tbl_dir: Directory containing tbl result files
|
||||
output_file: Output file path
|
||||
"""
|
||||
print("Starting file processing...")
|
||||
print(f"FASTA directory: {fasta_dir}")
|
||||
print(f"tbl directory: {tbl_dir}")
|
||||
print(f"Output file: {output_file}")
|
||||
print("-" * 50)
|
||||
|
||||
og_list = {}
|
||||
|
||||
# Find corresponding file pairs
|
||||
file_pairs = find_corresponding_files(fasta_dir, tbl_dir)
|
||||
|
||||
if not file_pairs:
|
||||
print("No corresponding FASTA-tbl file pairs found")
|
||||
return
|
||||
|
||||
print(f"Found {len(file_pairs)} file pairs to process")
|
||||
|
||||
# Process each file pair
|
||||
for i, file_pair in enumerate(file_pairs, 1):
|
||||
print(f"\nProcessing pair {i}/{len(file_pairs)}:")
|
||||
stem, seq_ids = add_best_hit_to_og_list(file_pair)
|
||||
if not seq_ids:
|
||||
print(f"Skipping {stem} due to no valid hits")
|
||||
continue
|
||||
og_list[stem] = seq_ids
|
||||
|
||||
with open(output_file, "w") as out_f:
|
||||
for og, ids in og_list.items():
|
||||
out_f.write(f"{og}\t" + "\t".join(ids) + "\n")
|
||||
|
||||
print(f"\nProcessing completed! All results saved to: {output_file}")
|
||||
|
||||
|
||||
def main(fasta_directory, tbl_directory, output_file):
|
||||
"""
|
||||
Main function - take paths and run processing
|
||||
|
||||
Args:
|
||||
fasta_directory: Folder containing FASTA files
|
||||
tbl_directory: Folder containing tbl result files
|
||||
output_file: Output file path
|
||||
"""
|
||||
# Check if input directories exist
|
||||
if not os.path.exists(fasta_directory):
|
||||
print(f"Error: FASTA directory does not exist {fasta_directory}")
|
||||
return
|
||||
|
||||
if not os.path.exists(tbl_directory):
|
||||
print(f"Error: tbl directory does not exist {tbl_directory}")
|
||||
return
|
||||
|
||||
# Process all files
|
||||
process_all_files(fasta_directory, tbl_directory, output_file)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Parse command-line arguments and call main
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Add HMMER best hit into orthologs FASTA."
|
||||
)
|
||||
parser.add_argument(
|
||||
"-f",
|
||||
"--fasta_directory",
|
||||
required=True,
|
||||
help="Directory containing FASTA files",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-t",
|
||||
"--tbl_directory",
|
||||
required=True,
|
||||
help="Directory containing tbl result files",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-o",
|
||||
"--output_file",
|
||||
required=True,
|
||||
help="File to write output seqname",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
main(args.fasta_directory, args.tbl_directory, args.output_file)
|
||||
Executable
+92
@@ -0,0 +1,92 @@
|
||||
#! /usr/bin/env python3
|
||||
import os
|
||||
import argparse
|
||||
from Bio import SeqIO
|
||||
from Bio.SeqRecord import SeqRecord
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def parse_fasta(fasta_file_path):
|
||||
"""
|
||||
Parse a FASTA file and return a list of SeqRecord.
|
||||
Args:
|
||||
fasta_file_path: Path to the FASTA file
|
||||
Returns:
|
||||
list: List of SeqRecord
|
||||
"""
|
||||
try:
|
||||
records = list(SeqIO.parse(fasta_file_path, "fasta"))
|
||||
return records
|
||||
except Exception as e:
|
||||
print(f"Error parsing FASTA file {fasta_file_path}: {e}")
|
||||
return []
|
||||
|
||||
|
||||
def ogs_to_fasta(ogs_name, seq_list, source_records, output_dir):
|
||||
"""
|
||||
Convert OGS list to FASTA format.
|
||||
|
||||
Args:
|
||||
ogs_name: Name of the OGS
|
||||
seq_list: List of sequence IDs
|
||||
source_records: List of SeqRecord from source FASTA
|
||||
output_dir: Directory to save the output FASTA file
|
||||
"""
|
||||
output_path = Path(output_dir) / f"{ogs_name}.fa"
|
||||
seq_dict = {record.id: record for record in source_records}
|
||||
|
||||
with open(output_path, "w") as out_f:
|
||||
for seq_id in seq_list:
|
||||
if seq_id in seq_dict:
|
||||
updated_id = seq_id.split("@")[0]
|
||||
updated_record = SeqRecord(
|
||||
seq_dict[seq_id].seq,
|
||||
id=updated_id,
|
||||
description="",
|
||||
)
|
||||
SeqIO.write(updated_record, out_f, "fasta")
|
||||
else:
|
||||
print(f"Warning: Sequence ID {seq_id} not found in source records.")
|
||||
|
||||
|
||||
def process_ogs_list(ogs_file, source_fasta, output_dir):
|
||||
"""
|
||||
Process OGS list file and convert each OGS to FASTA format.
|
||||
|
||||
Args:
|
||||
ogs_file: Path to the OGS list file
|
||||
source_fasta: Path to the source FASTA file
|
||||
output_dir: Directory to save the output FASTA files
|
||||
"""
|
||||
source_records = parse_fasta(source_fasta)
|
||||
|
||||
with open(ogs_file, "r") as f:
|
||||
for line in f:
|
||||
parts = line.strip().split("\t")
|
||||
if len(parts) < 2:
|
||||
continue
|
||||
ogs_name = parts.pop(0)
|
||||
ogs_to_fasta(ogs_name, parts, source_records, output_dir)
|
||||
print(f"Processed OGS: {ogs_name}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description="Convert OGS list to FASTA format.")
|
||||
parser.add_argument(
|
||||
"-i", "--input_ogs", required=True, help="Path to the OGS list file"
|
||||
)
|
||||
parser.add_argument(
|
||||
"-s", "--source_fasta", required=True, help="Path to the source FASTA file"
|
||||
)
|
||||
parser.add_argument(
|
||||
"-o",
|
||||
"--output_dir",
|
||||
required=True,
|
||||
help="Directory to save the output FASTA files",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
os.makedirs(args.output_dir, exist_ok=True)
|
||||
|
||||
process_ogs_list(args.input_ogs, args.source_fasta, args.output_dir)
|
||||
Executable
+226
@@ -0,0 +1,226 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
from xvfbwrapper import Xvfb
|
||||
import argparse
|
||||
import re
|
||||
import json
|
||||
from ete3 import Tree, TreeStyle, NodeStyle, faces
|
||||
|
||||
helptext= '''
|
||||
Generate the "Pie Chart" representation of gene tree conflict from Smith et al. 2015 from
|
||||
the output of phyparts, the bipartition summary software described in the same paper.
|
||||
|
||||
The input files include three files produced by PhyParts, and a file containing a species
|
||||
tree in Newick format (likely, the tree used for PhyParts). The output is an SVG containing
|
||||
the phylogeny along with pie charts at each node.
|
||||
|
||||
Requirements:
|
||||
|
||||
Python 3
|
||||
ete3
|
||||
|
||||
'''
|
||||
|
||||
|
||||
|
||||
vdisplay = Xvfb()
|
||||
vdisplay.start()
|
||||
|
||||
|
||||
|
||||
|
||||
#Read in species tree and convert to ultrametric
|
||||
|
||||
#Match phyparts nodes to ete3 nodes
|
||||
def get_phyparts_nodes(sptree_fn,phyparts_root):
|
||||
sptree = Tree(sptree_fn)
|
||||
sptree.convert_to_ultrametric()
|
||||
|
||||
phyparts_node_key = [line for line in open(phyparts_root+".node.key")]
|
||||
subtrees_dict = {n.split()[0]:Tree(n.split()[1]+";") for n in phyparts_node_key}
|
||||
subtrees_topids = {}
|
||||
for x in subtrees_dict:
|
||||
subtrees_topids[x] = subtrees_dict[x].get_topology_id()
|
||||
#print(subtrees_topids['1'])
|
||||
#print()
|
||||
for node in sptree.traverse():
|
||||
node_topid = node.get_topology_id()
|
||||
if "Takakia_4343a" in node.get_leaf_names():
|
||||
print(node_topid)
|
||||
print(node)
|
||||
for subtree in subtrees_dict:
|
||||
if node_topid == subtrees_topids[subtree]:
|
||||
node.name = subtree
|
||||
return sptree,subtrees_dict,subtrees_topids
|
||||
|
||||
#Summarize concordance and conflict from Phyparts
|
||||
def get_concord_and_conflict(phyparts_root,subtrees_dict,subtrees_topids):
|
||||
|
||||
with open(phyparts_root + ".concon.tre") as phyparts_trees:
|
||||
concon_tree = Tree(phyparts_trees.readline())
|
||||
conflict_tree = Tree(phyparts_trees.readline())
|
||||
|
||||
concord_dict = {}
|
||||
conflict_dict = {}
|
||||
|
||||
|
||||
for node in concon_tree.traverse():
|
||||
node_topid = node.get_topology_id()
|
||||
for subtree in subtrees_dict:
|
||||
if node_topid == subtrees_topids[subtree]:
|
||||
concord_dict[subtree] = node.support
|
||||
|
||||
for node in conflict_tree.traverse():
|
||||
node_topid = node.get_topology_id()
|
||||
for subtree in subtrees_dict:
|
||||
if node_topid == subtrees_topids[subtree]:
|
||||
conflict_dict[subtree] = node.support
|
||||
return concord_dict, conflict_dict
|
||||
|
||||
#Generate Pie Chart data
|
||||
def get_pie_chart_data(phyparts_root,total_genes,concord_dict,conflict_dict):
|
||||
|
||||
phyparts_hist = [line for line in open(phyparts_root + ".hist")]
|
||||
phyparts_pies = {}
|
||||
phyparts_dict = {}
|
||||
|
||||
for n in phyparts_hist:
|
||||
n = n.split(",")
|
||||
tot_genes = float(n.pop(-1))
|
||||
node_name = n.pop(0)[4:]
|
||||
concord = float(n.pop(0))
|
||||
concord = concord_dict[node_name]
|
||||
all_conflict = conflict_dict[node_name]
|
||||
|
||||
if len(n) > 0:
|
||||
most_conflict = max([float(x) for x in n])
|
||||
else:
|
||||
most_conflict = 0.0
|
||||
|
||||
adj_concord = (concord/total_genes) * 100
|
||||
adj_most_conflict = (most_conflict/total_genes) * 100
|
||||
other_conflict = (all_conflict - most_conflict) / total_genes * 100
|
||||
the_rest = (total_genes - concord - all_conflict) / total_genes * 100
|
||||
|
||||
pie_list = [adj_concord,adj_most_conflict,other_conflict,the_rest]
|
||||
|
||||
phyparts_pies[node_name] = pie_list
|
||||
|
||||
phyparts_dict[node_name] = [int(round(concord,0)),int(round(tot_genes-concord,0))]
|
||||
|
||||
return phyparts_dict, phyparts_pies
|
||||
|
||||
|
||||
def node_text_layout(mynode):
|
||||
F = faces.TextFace(mynode.name,fsize=20)
|
||||
faces.add_face_to_node(F,mynode,0,position="branch-right")
|
||||
|
||||
#convert internal phypartspiechart.py data files to csv and export to current directory (for use as ggtree tree data in R)
|
||||
def pie_data_to_csv(phyparts_dict, phyparts_pies):
|
||||
phyparts_dist_bin = {}
|
||||
phyparts_pies_bin = {}
|
||||
dist_replaced = {}
|
||||
pies_replaced = {}
|
||||
|
||||
phyparts_dist_bin = json.dumps(phyparts_dist)
|
||||
phyparts_pies_bin = json.dumps(phyparts_pies)
|
||||
|
||||
|
||||
dist_replaced = re.sub(r'{',r'node,concord,genes-concord\n',phyparts_dist_bin)
|
||||
dist_replaced = re.sub(r'"(\d*)":\s\[(\d*),\s(\d*)\],\s', r'\1,\2,\3\n', dist_replaced)
|
||||
dist_replaced = re.sub(r'"(\d*)":\s\[(\d*),\s(\d*)\]}', r'\1,\2,\3', dist_replaced)
|
||||
|
||||
pies_replaced = re.sub(r'{',r'node,adj_concord,adj_most_conflict,other_conflict,the_rest\n',phyparts_pies_bin)
|
||||
pies_replaced = re.sub(r'"(\d*)":\s\[(\d*.\d*),\s(\d*.\d*),\s(\d*.\d*),\s(\d*.\d*)\],\s', r'\1,\2,\3,\4,\5\n', pies_replaced)
|
||||
pies_replaced = re.sub(r'"(\d*)":\s\[(\d*.\d*),\s(\d*.\d*),\s(\d*.\d*),\s(\d*.\d*)\]}', r'\1,\2,\3,\4,\5', pies_replaced)
|
||||
|
||||
with open('phyparts_dist.csv','w') as file:
|
||||
for line in dist_replaced:
|
||||
file.write(line)
|
||||
with open('phyparts_pies.csv','w') as file:
|
||||
for line in pies_replaced:
|
||||
file.write(line)
|
||||
|
||||
|
||||
parser = argparse.ArgumentParser(description=helptext,formatter_class=argparse.RawTextHelpFormatter)
|
||||
parser.add_argument('species_tree',help="Newick formatted species tree topology.")
|
||||
parser.add_argument('phyparts_root',help="File root name used for Phyparts.")
|
||||
parser.add_argument('num_genes',type=int,default=0,help="Number of total gene trees. Used to properly scale pie charts.")
|
||||
parser.add_argument('--taxon_subst',help="Comma-delimted file to translate tip names.")
|
||||
parser.add_argument("--svg_name",help="File name for SVG generated by script",default="pies.svg")
|
||||
parser.add_argument("--show_nodes",help="Also show tree with nodes labeled same as PhyParts",action="store_true",default=False)
|
||||
parser.add_argument("--colors",help="Four colors of the pie chart: concordance (blue) top conflict (green), other conflict (red), no signal (gray)",nargs="+",default=["blue","green","red","dark gray"])
|
||||
parser.add_argument("--no_ladderize",help="Do not ladderize the input species tree.",action="store_true",default=False)
|
||||
parser.add_argument("--to_csv",help="Output data files to csv for import into ggtree in R",action="store_true",default=False)
|
||||
|
||||
args = parser.parse_args()
|
||||
if args.no_ladderize:
|
||||
ladderize=False
|
||||
else:
|
||||
ladderize=True
|
||||
plot_tree,subtrees_dict,subtrees_topids = get_phyparts_nodes(args.species_tree, args.phyparts_root)
|
||||
#print(subtrees_dict)
|
||||
concord_dict, conflict_dict = get_concord_and_conflict(args.phyparts_root,subtrees_dict,subtrees_topids)
|
||||
phyparts_dist, phyparts_pies = get_pie_chart_data(args.phyparts_root,args.num_genes,concord_dict,conflict_dict)
|
||||
|
||||
if args.taxon_subst:
|
||||
taxon_subst = {line.split(",")[0]:line.rstrip().split(",")[1] for line in open(args.taxon_subst,'U')}
|
||||
for leaf in plot_tree.get_leaves():
|
||||
try:
|
||||
leaf.name = taxon_subst[leaf.name]
|
||||
except KeyError:
|
||||
print(leaf.name)
|
||||
continue
|
||||
def phyparts_pie_layout(mynode):
|
||||
if mynode.name in phyparts_pies:
|
||||
pie= faces.PieChartFace(phyparts_pies[mynode.name],
|
||||
#colors=COLOR_SCHEMES["set1"],
|
||||
colors = args.colors,
|
||||
width=50, height=50)
|
||||
pie.border.width = None
|
||||
pie.opacity = 1
|
||||
faces.add_face_to_node(pie,mynode, 0, position="branch-right")
|
||||
|
||||
concord_text = faces.TextFace(str(int(concord_dict[mynode.name]))+' ',fsize=20)
|
||||
conflict_text = faces.TextFace(str(int(conflict_dict[mynode.name]))+' ',fsize=20)
|
||||
|
||||
faces.add_face_to_node(concord_text,mynode,0,position = "branch-top")
|
||||
faces.add_face_to_node(conflict_text,mynode,0,position="branch-bottom")
|
||||
|
||||
|
||||
else:
|
||||
F = faces.TextFace(mynode.name,fsize=20)
|
||||
faces.add_face_to_node(F,mynode,0,position="aligned")
|
||||
|
||||
#Plot Pie Chart
|
||||
ts = TreeStyle()
|
||||
ts.show_leaf_name = False
|
||||
|
||||
ts.layout_fn = phyparts_pie_layout
|
||||
nstyle = NodeStyle()
|
||||
nstyle["size"] = 0
|
||||
for n in plot_tree.traverse():
|
||||
n.set_style(nstyle)
|
||||
n.img_style["vt_line_width"] = 0
|
||||
|
||||
ts.draw_guiding_lines = True
|
||||
ts.guiding_lines_color = "black"
|
||||
ts.guiding_lines_type = 0
|
||||
ts.scale = 30
|
||||
ts.branch_vertical_margin = 10
|
||||
plot_tree.convert_to_ultrametric()
|
||||
if args.to_csv:
|
||||
pie_data_to_csv(phyparts_dist, phyparts_pies)
|
||||
|
||||
if ladderize:
|
||||
plot_tree.ladderize(direction=1)
|
||||
my_svg = plot_tree.render(args.svg_name,tree_style=ts,w=595,dpi=300)
|
||||
|
||||
if args.show_nodes:
|
||||
node_style = TreeStyle()
|
||||
node_style.show_leaf_name=False
|
||||
node_style.layout_fn = node_text_layout
|
||||
plot_tree.render("tree_nodes.pdf",tree_style=node_style)
|
||||
|
||||
vdisplay.stop()
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
#!/usr/bin/env python3
|
||||
"samtools consensus -r NC_058887.1:11845-11988 --show-del yes -A -H 0.3 -m simple --show-ins yes -f fasta EP.sorted.bam"
|
||||
Executable
+71
@@ -0,0 +1,71 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Trinity FASTA Sequence Renaming Script
|
||||
Function: Rename sequences in FASTA file to format: [prefix@sequence_number]
|
||||
"""
|
||||
|
||||
import sys
|
||||
import os
|
||||
|
||||
def rename_fasta_sequences(input_file, prefix, output_file=None):
|
||||
"""
|
||||
Rename sequence headers in a FASTA file
|
||||
|
||||
Parameters:
|
||||
input_file: Input FASTA filename
|
||||
prefix: Prefix for sequence names
|
||||
output_file: Output filename (optional, defaults to input_file_renamed.fasta)
|
||||
"""
|
||||
|
||||
# Set output filename
|
||||
if output_file is None:
|
||||
file_base, file_ext = os.path.splitext(input_file)
|
||||
output_file = f"{file_base}_renamed{file_ext}"
|
||||
|
||||
# Counter for sequences
|
||||
seq_count = 0
|
||||
|
||||
try:
|
||||
with open(input_file, 'r') as fin, open(output_file, 'w') as fout:
|
||||
for line in fin:
|
||||
if line.startswith('>'):
|
||||
# Sequence header line: rename it
|
||||
seq_count += 1
|
||||
new_name = f">{prefix}@mrna_{seq_count}\n"
|
||||
fout.write(new_name)
|
||||
else:
|
||||
# Sequence data line: write as-is
|
||||
fout.write(line)
|
||||
|
||||
print(f"Successfully renamed {seq_count} sequences")
|
||||
print(f"Input file: {input_file}")
|
||||
print(f"Output file: {output_file}")
|
||||
print(f"Naming format: {prefix}@mrna_number")
|
||||
|
||||
except FileNotFoundError:
|
||||
print(f"Error: Input file '{input_file}' not found")
|
||||
sys.exit(1)
|
||||
except Exception as e:
|
||||
print(f"Error processing file: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
def main():
|
||||
"""Main function"""
|
||||
if len(sys.argv) < 3:
|
||||
print("Usage: python script.py <fasta_file> <prefix> [output_file]")
|
||||
print("Example: python script.py sequences.fasta Gene new_sequences.fasta")
|
||||
sys.exit(1)
|
||||
|
||||
input_file = sys.argv[1]
|
||||
prefix = sys.argv[2]
|
||||
output_file = sys.argv[3] if len(sys.argv) > 3 else None
|
||||
|
||||
# Verify input file exists
|
||||
if not os.path.isfile(input_file):
|
||||
print(f"Error: File '{input_file}' does not exist")
|
||||
sys.exit(1)
|
||||
|
||||
rename_fasta_sequences(input_file, prefix, output_file)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user