20251125
This commit is contained in:
@@ -0,0 +1,7 @@
|
||||
#! /usr/bin/env bash
|
||||
mkdir -p modeltests
|
||||
for i in ../gene_alignment/*.fa ; do
|
||||
j=$(basename "$i")
|
||||
echo "modeltest-ng -p 2 -r 12345 --force -i $i -d nt -t ml -o modeltests/${j/.fa/}.modeltest" >> modeltest.cmds
|
||||
done
|
||||
xargs -t -P 4 -I cmd -a modeltest.cmds bash -c "cmd"
|
||||
@@ -0,0 +1,9 @@
|
||||
#! /usr/bin/env bash
|
||||
mkdir -p raxml_ng
|
||||
echo -n > raxml_ng.cmds
|
||||
for i in modeltests/*.modeltest.out ; do
|
||||
j=$(basename "$i")
|
||||
cmd=$(grep "raxml-ng" "$i" | tail -n 1 | sed 's/> //')
|
||||
echo "$cmd --all --bs-trees 1000 --outgroup Zju --redo --threads 4 --seed 12345 --prefix raxml_ng/${j/.modeltest.out/} > /dev/null" >> raxml_ng.cmds
|
||||
done
|
||||
xargs -t -P 3 -I cmd -a raxml_ng.cmds bash -c "cmd"
|
||||
@@ -0,0 +1,89 @@
|
||||
#! /usr/bin/env julia
|
||||
## Installing Dependencies
|
||||
## using Pkg
|
||||
## Pkg.add("Distributed")
|
||||
## Pkg.add("DataFrames")
|
||||
## Pkg.add("CSV")
|
||||
## Pkg.add("SNaQ")
|
||||
## Pkg.add("PhyloNetworks")
|
||||
## Pkg.add("RCall")
|
||||
## Pkg.add("PhyloPlots")
|
||||
## Pkg.add("QuartetNetworkGoodnessFit")
|
||||
|
||||
# Running SNaQ Analysis
|
||||
using PhyloNetworks, SNaQ;
|
||||
using Distributed;
|
||||
addprocs(5);
|
||||
@everywhere using PhyloNetworks, SNaQ;
|
||||
nruns = 100; # number of runs for each hmax
|
||||
astralfile = joinpath("..", "..", "species_tree", "aster.out");
|
||||
astraltree = readnewick(astralfile);
|
||||
|
||||
### Reading RAxML gene trees and ASTRAL species tree
|
||||
### running in raxml_snaq/ folder
|
||||
# raxmltrees = joinpath("..", "..", "species_tree", "all.trees");
|
||||
# inputCF = readtrees2CF(raxmltrees);
|
||||
# net0 = snaq!(astraltree, inputCF, hmax=0, filename="net0", seed=123, outgroup="Zju", runs=nruns);
|
||||
# net1 = snaq!(net0, inputCF, hmax=1, filename="net1", seed=123, outgroup="Zju", runs=nruns);
|
||||
# net2 = snaq!(net1, inputCF, hmax=2, filename="net2", seed=123, outgroup="Zju", runs=nruns);
|
||||
# net3 = snaq!(net2, inputCF, hmax=3, filename="net3", seed=123, outgroup="Zju", runs=nruns);
|
||||
# net4 = snaq!(net3, inputCF, hmax=4, filename="net4", seed=123, outgroup="Zju", runs=nruns);
|
||||
|
||||
### Alternatively, reading in the input files from Bucky
|
||||
### running in input_snaq/ folder
|
||||
inputCFfile = joinpath("..", "..", "input", "input.CFs.csv");
|
||||
inputCF = readtableCF(inputCFfile);
|
||||
net0 = snaq!(astraltree, inputCF, hmax=0, filename="net0", seed=123, outgroup="Zju", runs=nruns);
|
||||
net1 = snaq!(net0, inputCF, hmax=1, filename="net1", seed=123, outgroup="Zju", runs=nruns);
|
||||
net2 = snaq!(net1, inputCF, hmax=2, filename="net2", seed=123, outgroup="Zju", runs=nruns);
|
||||
net3 = snaq!(net2, inputCF, hmax=3, filename="net3", seed=123, outgroup="Zju", runs=nruns);
|
||||
net4 = snaq!(net3, inputCF, hmax=4, filename="net4", seed=123, outgroup="Zju", runs=nruns);
|
||||
|
||||
|
||||
# Plotting the SNaQ results
|
||||
using PhyloPlots, RCall;
|
||||
## Network scores vs. hmax
|
||||
scores = [loglik(net0), loglik(net1), loglik(net2), loglik(net3), loglik(net4)];
|
||||
hmax = collect(0:4);
|
||||
R"pdf"("snaq_network_scores.pdf", width=12, height=8);
|
||||
R"plot"(hmax, scores, type="b", ylab="network score", xlab="hmax", col="blue");
|
||||
R"dev.off"();
|
||||
|
||||
## Rerooting and rotating the networks for better visualization
|
||||
rootatnode!(net1, "Zju");
|
||||
rootatnode!(net2, "Zju");
|
||||
rootatnode!(net3, "Zju");
|
||||
rootatnode!(net4, "Zju");
|
||||
### rotate!(net1, -2);
|
||||
### rotate!(net2, -2);
|
||||
### rotate!(net3, -2);
|
||||
### rotate!(net4, -2);
|
||||
|
||||
## Plotting the networks
|
||||
R"pdf"("snaq_networks.pdf", width=14, height=10);
|
||||
R"layout(matrix(1:4, 2, 2, byrow=TRUE))"; # to get 4 plots into a single figure: 2 row, 2 columns
|
||||
R"par"(mar=[0, 0, 1.5, 0]); # for smaller margins
|
||||
xmin, xmax = PhyloPlots.PhyloPlots.edgenode_coordinates(net1, false, false)[13:14];
|
||||
xmax += (xmax - xmin) * 0.3;
|
||||
plot(net1, showgamma=true, tipoffset=0.1, xlim=[xmin, xmax]);
|
||||
R"mtext"(string("hmax=1, loglik=-", round(loglik(net1), digits=2)), font=2);
|
||||
xmin, xmax = PhyloPlots.PhyloPlots.edgenode_coordinates(net2, false, false)[13:14];
|
||||
xmax += (xmax - xmin) * 0.3;
|
||||
plot(net2, showgamma=true, tipoffset=0.1, xlim=[xmin, xmax]);
|
||||
R"mtext"(string("hmax=2, loglik=-", round(loglik(net2), digits=2)), font=2);
|
||||
xmin, xmax = PhyloPlots.PhyloPlots.edgenode_coordinates(net3, false, false)[13:14];
|
||||
xmax += (xmax - xmin) * 0.3;
|
||||
plot(net3, showgamma=true, tipoffset=0.1, xlim=[xmin, xmax]);
|
||||
R"mtext"(string("hmax=3, loglik=-", round(loglik(net3), digits=2)), font=2);
|
||||
xmin, xmax = PhyloPlots.PhyloPlots.edgenode_coordinates(net4, false, false)[13:14];
|
||||
xmax += (xmax - xmin) * 0.3;
|
||||
plot(net4, showgamma=true, tipoffset=0.1, xlim=[xmin, xmax]);
|
||||
R"mtext"(string("hmax=4, loglik=-", round(loglik(net4), digits=2)), font=2);
|
||||
R"dev.off"();
|
||||
|
||||
## expected vs. observed quartet concordance factors
|
||||
using CSV, DataFrames;
|
||||
|
||||
# Goodness of fit of the SNaQ networks
|
||||
using QuartetNetworkGoodnessFit;
|
||||
|
||||
@@ -0,0 +1,146 @@
|
||||
#! /usr/bin/env python3
|
||||
import os
|
||||
from Bio import SeqIO
|
||||
from Bio.Seq import Seq
|
||||
from Bio.SeqRecord import SeqRecord
|
||||
from collections import defaultdict
|
||||
import argparse
|
||||
|
||||
|
||||
def get_sequence_lengths(fasta_files):
|
||||
"""
|
||||
get the lengths of sequences in each FASTA file
|
||||
Assumes all sequences in a file have the same length
|
||||
"""
|
||||
file_lengths = {}
|
||||
for fasta_file in fasta_files:
|
||||
try:
|
||||
with open(fasta_file, "r") as f:
|
||||
for record in SeqIO.parse(f, "fasta"):
|
||||
# get length of the first sequence
|
||||
file_lengths[fasta_file] = len(record.seq)
|
||||
break
|
||||
except Exception as e:
|
||||
print(f"Error reading file {fasta_file}: {e}")
|
||||
file_lengths[fasta_file] = 0
|
||||
|
||||
return file_lengths
|
||||
|
||||
|
||||
def concatenate_fasta_files(fasta_files, output_file):
|
||||
"""
|
||||
Concatenate sequences from multiple FASTA files by name, using "-" for missing sequences.
|
||||
"""
|
||||
# Get the sequence lengths for each file
|
||||
file_lengths = get_sequence_lengths(fasta_files)
|
||||
|
||||
# Store all sequence names and their corresponding content
|
||||
sequences_dict = defaultdict(dict)
|
||||
all_sequence_names = set()
|
||||
|
||||
# Read sequences from each file
|
||||
for i, fasta_file in enumerate(fasta_files):
|
||||
try:
|
||||
with open(fasta_file, "r") as f:
|
||||
for record in SeqIO.parse(f, "fasta"):
|
||||
seq_name = record.id
|
||||
sequences_dict[seq_name][i] = str(record.seq)
|
||||
all_sequence_names.add(seq_name)
|
||||
except Exception as e:
|
||||
print(f"Error reading file {fasta_file}: {e}")
|
||||
|
||||
# Create concatenated sequences
|
||||
concatenated_sequences = []
|
||||
|
||||
for seq_name in sorted(all_sequence_names):
|
||||
concatenated_seq = []
|
||||
|
||||
for i, fasta_file in enumerate(fasta_files):
|
||||
if i in sequences_dict[seq_name]:
|
||||
# This file has the sequence, add it directly
|
||||
concatenated_seq.append(sequences_dict[seq_name][i])
|
||||
else:
|
||||
# This file is missing the sequence, use "-" to fill the gap
|
||||
gap_length = file_lengths[fasta_file]
|
||||
concatenated_seq.append("-" * gap_length)
|
||||
|
||||
# Concatenate all parts of the sequence
|
||||
full_sequence = "".join(concatenated_seq)
|
||||
|
||||
# Create a new sequence record
|
||||
|
||||
new_record = SeqRecord(
|
||||
Seq(full_sequence),
|
||||
id=seq_name,
|
||||
description=f"concatenated_from_{len(fasta_files)}_files",
|
||||
)
|
||||
concatenated_sequences.append(new_record)
|
||||
|
||||
with open(output_file, "w") as output_handle:
|
||||
SeqIO.write(concatenated_sequences, output_handle, "fasta")
|
||||
|
||||
print(
|
||||
f"Successfully concatenate {len(concatenated_sequences)} sequences to {output_file}"
|
||||
)
|
||||
print(f"Input file count: {len(fasta_files)}")
|
||||
|
||||
# Output statistics
|
||||
for i, fasta_file in enumerate(fasta_files):
|
||||
seq_count = sum(1 for seqs in sequences_dict.values() if i in seqs)
|
||||
print(
|
||||
f"File {i + 1}: {os.path.basename(fasta_file)} - Sequence count: {seq_count}, Sequence length: {file_lengths[fasta_file]}."
|
||||
)
|
||||
|
||||
print(f"Total output sequence length: {len(concatenated_sequences[0].seq)}.")
|
||||
|
||||
|
||||
def get_fasta_files_from_directory(directory, extensions):
|
||||
"""
|
||||
get all FASTA files from a directory with specified extensions
|
||||
"""
|
||||
fasta_files = []
|
||||
for filename in os.listdir(directory):
|
||||
if any(filename.endswith(ext) for ext in extensions):
|
||||
fasta_files.append(os.path.join(directory, filename))
|
||||
return sorted(fasta_files)
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Concatenate multiple FASTA files by sequence names."
|
||||
)
|
||||
parser.add_argument("-i", "--input", nargs="+", help="Input FASTA file list")
|
||||
parser.add_argument("-d", "--directory", help="Directory containing FASTA files")
|
||||
parser.add_argument("-o", "--output", required=True, help="Output file")
|
||||
parser.add_argument(
|
||||
"-e",
|
||||
"--extensions",
|
||||
nargs="+",
|
||||
default=[".fasta", ".fa", ".fna"],
|
||||
help="FASTA file extensions to look for in directory",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# 获取输入文件
|
||||
if args.directory:
|
||||
fasta_files = get_fasta_files_from_directory(args.directory, args.extensions)
|
||||
if not fasta_files:
|
||||
print(
|
||||
f"Cannot find FASTA files in {args.directory} with extensions {args.extensions}"
|
||||
)
|
||||
return
|
||||
elif args.input:
|
||||
fasta_files = args.input
|
||||
else:
|
||||
print("Please specify input files or directory")
|
||||
return
|
||||
|
||||
print(f"Found {len(fasta_files)} FASTA files:")
|
||||
|
||||
# Perform concatenation
|
||||
concatenate_fasta_files(fasta_files, args.output)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,26 @@
|
||||
#! /usr/bin/env Rscript
|
||||
# DensiTree visualization of phylogenetic trees
|
||||
args <- commandArgs(trailingOnly = TRUE)
|
||||
if (length(args) != 6) {
|
||||
stop("Usage: Rscript 05.densitree.r <tree_file> <tip_order_file> <root> <output_pdf> <width> <height>")
|
||||
}
|
||||
tree_file <- args[1]
|
||||
tip_order_file <- args[2]
|
||||
root <- args[3]
|
||||
output_pdf <- args[4]
|
||||
width <- as.numeric(args[5])
|
||||
height <- as.numeric(args[6])
|
||||
|
||||
library(ape)
|
||||
library(phangorn)
|
||||
trees <- read.tree(tree_file)
|
||||
pdf(output_pdf, width = width, height = height)
|
||||
for(i in 1:length(trees)) {
|
||||
trees[[i]] <- compute.brlen(root(trees[[i]], root))
|
||||
}
|
||||
tip_order <- rev(readLines(tip_order_file))
|
||||
densiTree(trees, consensus=tip_order, alpha = 0.01,
|
||||
col = "#009900", type = "cladogram",
|
||||
label.offset = 0.02, scale.bar = FALSE
|
||||
)
|
||||
dev.off()
|
||||
@@ -0,0 +1,18 @@
|
||||
#! /usr/bin/env bash
|
||||
|
||||
if [ "$#" -ne 3 ]; then
|
||||
echo "Usage: $0 <input_fasta_dir> <extension> <output_nexus_dir>"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
input_dir=$1
|
||||
extension=$2
|
||||
output_dir=$3
|
||||
mkdir -p "${output_dir}"
|
||||
for f in "${input_dir}"/*."${extension}"; do
|
||||
filename=$(basename -- "${f}")
|
||||
filename_noext="${filename%.*}"
|
||||
output_file="${output_dir}/${filename_noext}.nex"
|
||||
echo "Converting ${f} to ${output_file}"
|
||||
seqmagick convert --output-format nexus --alphabet dna --input-format fasta "${f}" "${output_file}"
|
||||
done
|
||||
@@ -0,0 +1,21 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
mkdir -p ../mbsum_out
|
||||
echo -n "" > ../mbsum.log
|
||||
for i in *.nex.tar.gz; do
|
||||
base=$(basename "$i" .nex.tar.gz)
|
||||
echo "Processing ${base}" >> ../mbsum.log
|
||||
mkdir -p "${base}"
|
||||
tar -xzf "$i" -C "${base}"
|
||||
## skip Average standard deviation of split frequencies > 0.01
|
||||
dsf=$(awk '/Average standard deviation of split frequencies:/ {out=$7} END{print out+0}' "${base}"/*.nex.log 2>/dev/null)
|
||||
if awk -v d="$dsf" 'BEGIN{if (d >= 0.01) exit 0; exit 1}'; then
|
||||
echo "Skipping ${base} due to high DSF: ${dsf}" >> ../mbsum.log
|
||||
continue
|
||||
else
|
||||
mbsum "${base}"/*.t -n 1000 -o ../mbsum_out/"${base}".in >> ../mbsum.log 2>&1
|
||||
fi
|
||||
rm -rf "${base}"
|
||||
echo "Completed ${base}" >> ../mbsum.log
|
||||
done
|
||||
echo "All done!"
|
||||
@@ -0,0 +1,12 @@
|
||||
begin mrbayes;
|
||||
set nowarnings=yes;
|
||||
set usebeagle=yes;
|
||||
set autoclose=yes;
|
||||
set seed=12345;
|
||||
set swapseed=12345;
|
||||
lset nst=6 rates=gamma;
|
||||
mcmcp ngen=2000000 burninfrac=.25 samplefreq=1000 printfreq=10000 checkpoint=no
|
||||
diagnfreq=10000 nruns=3 nchains=3 temp=0.40 swapfreq=10 stoprule=no;
|
||||
mcmc;
|
||||
sumt;
|
||||
end;
|
||||
Reference in New Issue
Block a user