This commit is contained in:
2025-11-25 00:28:51 +08:00
commit eb3f16c30e
406 changed files with 91653 additions and 0 deletions
@@ -0,0 +1,8 @@
#! /usr/bin/env bash
mkdir -p msa
echo -n > mafft.cmds
for i in ogs/*.fa ; do
j=$(basename "$i")
echo "linsi --quiet $i > msa/$j" >> mafft.cmds
done
xargs -t -P 8 -I cmd -a mafft.cmds bash -c "cmd"
@@ -0,0 +1,8 @@
#! /usr/bin/env bash
mkdir -p hmms
echo -n > hmmbuild.cmds
for i in msa/*.fa ; do
j=$(basename "$i")
echo "hmmbuild -o hmms/${j}.hmmbuild.out --amino hmms/${j}.hmm $i" >> hmmbuild.cmds
done
xargs -t -P 8 -I cmd -a hmmbuild.cmds bash -c "cmd"
@@ -0,0 +1,8 @@
#! /usr/bin/env bash
mkdir -p hmmsearch
echo -n > hmmsearch.cmds
for i in hmms/*.hmm ; do
j=$(basename "$i")
echo "hmmsearch --tblout hmmsearch/${j}search.tblout $i ../../01.reference/Zju.pep.fa > hmmsearch/${j}search.rawout" >> hmmsearch.cmds
done
xargs -t -P 8 -I cmd -a hmmsearch.cmds bash -c "cmd"
+8
View File
@@ -0,0 +1,8 @@
#! /usr/bin/env bash
mkdir -p pep_aln
echo -n > mafft.cmds
for i in raw_ogs/pep/*.fa; do
j=$(basename "$i")
echo "linsi --quiet $i > pep_aln/${j/.fa/.pal}" >> mafft.cmds
done
xargs -t -P 8 -I cmd -a mafft.cmds bash -c "cmd"
+8
View File
@@ -0,0 +1,8 @@
#! /usr/bin/env bash
mkdir -p cds_aln
echo -n > pal2nal.cmds
for i in pep_aln/*.pal; do
j=$(basename "$i")
echo "pal2nal.pl $i raw_ogs/cds/${j/.pal/.fa} -output fasta > cds_aln/${j/.pal/.nal}" >> pal2nal.cmds
done
xargs -t -P 8 -I cmd -a pal2nal.cmds bash -c "cmd"
@@ -0,0 +1,8 @@
#! /usr/bin/env bash
mkdir -p trimed_nal
echo -n > trimal.cmds
for i in cds_aln/*.nal ;do
j=$(basename "$i")
echo "trimal -in $i -out trimed_nal/${j/.nal/.trimed.fa} -automated1 -resoverlap 0.5 -seqoverlap 50" >> trimal.cmds
done
xargs -t -P 4 -I cmd -a trimal.cmds bash -c "cmd"
+8
View File
@@ -0,0 +1,8 @@
#! /usr/bin/env bash
mkdir -p fasttree
echo -n > fasttree.cmds
for i in trimed_nal/*.trimed.fa ;do
j=$(basename "$i")
echo "FastTree -nt -gtr -quiet $i > fasttree/${j/.trimed.fa/.tree}" >> fasttree.cmds
done
xargs -t -P 8 -I cmd -a fasttree.cmds bash -c "cmd"
+11
View File
@@ -0,0 +1,11 @@
#! /usr/bin/env bash
mkdir -p treeshrink
for i in trimed_nal/*.trimed.fa; do
j=$(basename "$i")
mkdir -p treeshrink/"${j/.trimed.fa/}"
cd treeshrink/"${j/.trimed.fa/}" || exit 1
ln -s ../../fasttree/"${j/.trimed.fa/.tree}" input.tree
ln -s ../../"$i" input.fasta
cd ../../
done
run_treeshrink.py -i treeshrink/ -t input.tree -a input.fasta > treeshrink.log
+12
View File
@@ -0,0 +1,12 @@
#! /usr/bin/env bash
total_taxon=11
min_seq_length=300
mkdir -p final_ogs
for i in treeshrink/* ; do
j=$(basename "$i")
seqlen=$(seqkit fx2tab -C ATCG "$i"/output.fasta | awk '{print $3}' | sort -n | head -n 1)
seqnum=$(grep -c ">" "$i"/output.fasta)
if [[ $seqnum -eq $total_taxon && $seqlen -ge $min_seq_length ]]; then
cp -l "$i"/output.fasta final_ogs/"${j}.fa"
fi
done
@@ -0,0 +1,7 @@
#! /usr/bin/env bash
mkdir -p modeltests
for i in ../gene_alignment/*.fa ; do
j=$(basename "$i")
echo "modeltest-ng -p 2 -r 12345 --force -i $i -d nt -t ml -o modeltests/${j/.fa/}.modeltest" >> modeltest.cmds
done
xargs -t -P 4 -I cmd -a modeltest.cmds bash -c "cmd"
@@ -0,0 +1,9 @@
#! /usr/bin/env bash
mkdir -p raxml_ng
echo -n > raxml_ng.cmds
for i in modeltests/*.modeltest.out ; do
j=$(basename "$i")
cmd=$(grep "raxml-ng" "$i" | tail -n 1 | sed 's/> //')
echo "$cmd --all --bs-trees 1000 --outgroup Zju --redo --threads 4 --seed 12345 --prefix raxml_ng/${j/.modeltest.out/} > /dev/null" >> raxml_ng.cmds
done
xargs -t -P 3 -I cmd -a raxml_ng.cmds bash -c "cmd"
@@ -0,0 +1,89 @@
#! /usr/bin/env julia
## Installing Dependencies
## using Pkg
## Pkg.add("Distributed")
## Pkg.add("DataFrames")
## Pkg.add("CSV")
## Pkg.add("SNaQ")
## Pkg.add("PhyloNetworks")
## Pkg.add("RCall")
## Pkg.add("PhyloPlots")
## Pkg.add("QuartetNetworkGoodnessFit")
# Running SNaQ Analysis
using PhyloNetworks, SNaQ;
using Distributed;
addprocs(5);
@everywhere using PhyloNetworks, SNaQ;
nruns = 100; # number of runs for each hmax
astralfile = joinpath("..", "..", "species_tree", "aster.out");
astraltree = readnewick(astralfile);
### Reading RAxML gene trees and ASTRAL species tree
### running in raxml_snaq/ folder
# raxmltrees = joinpath("..", "..", "species_tree", "all.trees");
# inputCF = readtrees2CF(raxmltrees);
# net0 = snaq!(astraltree, inputCF, hmax=0, filename="net0", seed=123, outgroup="Zju", runs=nruns);
# net1 = snaq!(net0, inputCF, hmax=1, filename="net1", seed=123, outgroup="Zju", runs=nruns);
# net2 = snaq!(net1, inputCF, hmax=2, filename="net2", seed=123, outgroup="Zju", runs=nruns);
# net3 = snaq!(net2, inputCF, hmax=3, filename="net3", seed=123, outgroup="Zju", runs=nruns);
# net4 = snaq!(net3, inputCF, hmax=4, filename="net4", seed=123, outgroup="Zju", runs=nruns);
### Alternatively, reading in the input files from Bucky
### running in input_snaq/ folder
inputCFfile = joinpath("..", "..", "input", "input.CFs.csv");
inputCF = readtableCF(inputCFfile);
net0 = snaq!(astraltree, inputCF, hmax=0, filename="net0", seed=123, outgroup="Zju", runs=nruns);
net1 = snaq!(net0, inputCF, hmax=1, filename="net1", seed=123, outgroup="Zju", runs=nruns);
net2 = snaq!(net1, inputCF, hmax=2, filename="net2", seed=123, outgroup="Zju", runs=nruns);
net3 = snaq!(net2, inputCF, hmax=3, filename="net3", seed=123, outgroup="Zju", runs=nruns);
net4 = snaq!(net3, inputCF, hmax=4, filename="net4", seed=123, outgroup="Zju", runs=nruns);
# Plotting the SNaQ results
using PhyloPlots, RCall;
## Network scores vs. hmax
scores = [loglik(net0), loglik(net1), loglik(net2), loglik(net3), loglik(net4)];
hmax = collect(0:4);
R"pdf"("snaq_network_scores.pdf", width=12, height=8);
R"plot"(hmax, scores, type="b", ylab="network score", xlab="hmax", col="blue");
R"dev.off"();
## Rerooting and rotating the networks for better visualization
rootatnode!(net1, "Zju");
rootatnode!(net2, "Zju");
rootatnode!(net3, "Zju");
rootatnode!(net4, "Zju");
### rotate!(net1, -2);
### rotate!(net2, -2);
### rotate!(net3, -2);
### rotate!(net4, -2);
## Plotting the networks
R"pdf"("snaq_networks.pdf", width=14, height=10);
R"layout(matrix(1:4, 2, 2, byrow=TRUE))"; # to get 4 plots into a single figure: 2 row, 2 columns
R"par"(mar=[0, 0, 1.5, 0]); # for smaller margins
xmin, xmax = PhyloPlots.PhyloPlots.edgenode_coordinates(net1, false, false)[13:14];
xmax += (xmax - xmin) * 0.3;
plot(net1, showgamma=true, tipoffset=0.1, xlim=[xmin, xmax]);
R"mtext"(string("hmax=1, loglik=-", round(loglik(net1), digits=2)), font=2);
xmin, xmax = PhyloPlots.PhyloPlots.edgenode_coordinates(net2, false, false)[13:14];
xmax += (xmax - xmin) * 0.3;
plot(net2, showgamma=true, tipoffset=0.1, xlim=[xmin, xmax]);
R"mtext"(string("hmax=2, loglik=-", round(loglik(net2), digits=2)), font=2);
xmin, xmax = PhyloPlots.PhyloPlots.edgenode_coordinates(net3, false, false)[13:14];
xmax += (xmax - xmin) * 0.3;
plot(net3, showgamma=true, tipoffset=0.1, xlim=[xmin, xmax]);
R"mtext"(string("hmax=3, loglik=-", round(loglik(net3), digits=2)), font=2);
xmin, xmax = PhyloPlots.PhyloPlots.edgenode_coordinates(net4, false, false)[13:14];
xmax += (xmax - xmin) * 0.3;
plot(net4, showgamma=true, tipoffset=0.1, xlim=[xmin, xmax]);
R"mtext"(string("hmax=4, loglik=-", round(loglik(net4), digits=2)), font=2);
R"dev.off"();
## expected vs. observed quartet concordance factors
using CSV, DataFrames;
# Goodness of fit of the SNaQ networks
using QuartetNetworkGoodnessFit;
@@ -0,0 +1,146 @@
#! /usr/bin/env python3
import os
from Bio import SeqIO
from Bio.Seq import Seq
from Bio.SeqRecord import SeqRecord
from collections import defaultdict
import argparse
def get_sequence_lengths(fasta_files):
"""
get the lengths of sequences in each FASTA file
Assumes all sequences in a file have the same length
"""
file_lengths = {}
for fasta_file in fasta_files:
try:
with open(fasta_file, "r") as f:
for record in SeqIO.parse(f, "fasta"):
# get length of the first sequence
file_lengths[fasta_file] = len(record.seq)
break
except Exception as e:
print(f"Error reading file {fasta_file}: {e}")
file_lengths[fasta_file] = 0
return file_lengths
def concatenate_fasta_files(fasta_files, output_file):
"""
Concatenate sequences from multiple FASTA files by name, using "-" for missing sequences.
"""
# Get the sequence lengths for each file
file_lengths = get_sequence_lengths(fasta_files)
# Store all sequence names and their corresponding content
sequences_dict = defaultdict(dict)
all_sequence_names = set()
# Read sequences from each file
for i, fasta_file in enumerate(fasta_files):
try:
with open(fasta_file, "r") as f:
for record in SeqIO.parse(f, "fasta"):
seq_name = record.id
sequences_dict[seq_name][i] = str(record.seq)
all_sequence_names.add(seq_name)
except Exception as e:
print(f"Error reading file {fasta_file}: {e}")
# Create concatenated sequences
concatenated_sequences = []
for seq_name in sorted(all_sequence_names):
concatenated_seq = []
for i, fasta_file in enumerate(fasta_files):
if i in sequences_dict[seq_name]:
# This file has the sequence, add it directly
concatenated_seq.append(sequences_dict[seq_name][i])
else:
# This file is missing the sequence, use "-" to fill the gap
gap_length = file_lengths[fasta_file]
concatenated_seq.append("-" * gap_length)
# Concatenate all parts of the sequence
full_sequence = "".join(concatenated_seq)
# Create a new sequence record
new_record = SeqRecord(
Seq(full_sequence),
id=seq_name,
description=f"concatenated_from_{len(fasta_files)}_files",
)
concatenated_sequences.append(new_record)
with open(output_file, "w") as output_handle:
SeqIO.write(concatenated_sequences, output_handle, "fasta")
print(
f"Successfully concatenate {len(concatenated_sequences)} sequences to {output_file}"
)
print(f"Input file count: {len(fasta_files)}")
# Output statistics
for i, fasta_file in enumerate(fasta_files):
seq_count = sum(1 for seqs in sequences_dict.values() if i in seqs)
print(
f"File {i + 1}: {os.path.basename(fasta_file)} - Sequence count: {seq_count}, Sequence length: {file_lengths[fasta_file]}."
)
print(f"Total output sequence length: {len(concatenated_sequences[0].seq)}.")
def get_fasta_files_from_directory(directory, extensions):
"""
get all FASTA files from a directory with specified extensions
"""
fasta_files = []
for filename in os.listdir(directory):
if any(filename.endswith(ext) for ext in extensions):
fasta_files.append(os.path.join(directory, filename))
return sorted(fasta_files)
def main():
parser = argparse.ArgumentParser(
description="Concatenate multiple FASTA files by sequence names."
)
parser.add_argument("-i", "--input", nargs="+", help="Input FASTA file list")
parser.add_argument("-d", "--directory", help="Directory containing FASTA files")
parser.add_argument("-o", "--output", required=True, help="Output file")
parser.add_argument(
"-e",
"--extensions",
nargs="+",
default=[".fasta", ".fa", ".fna"],
help="FASTA file extensions to look for in directory",
)
args = parser.parse_args()
# 获取输入文件
if args.directory:
fasta_files = get_fasta_files_from_directory(args.directory, args.extensions)
if not fasta_files:
print(
f"Cannot find FASTA files in {args.directory} with extensions {args.extensions}"
)
return
elif args.input:
fasta_files = args.input
else:
print("Please specify input files or directory")
return
print(f"Found {len(fasta_files)} FASTA files:")
# Perform concatenation
concatenate_fasta_files(fasta_files, args.output)
if __name__ == "__main__":
main()
@@ -0,0 +1,26 @@
#! /usr/bin/env Rscript
# DensiTree visualization of phylogenetic trees
args <- commandArgs(trailingOnly = TRUE)
if (length(args) != 6) {
stop("Usage: Rscript 05.densitree.r <tree_file> <tip_order_file> <root> <output_pdf> <width> <height>")
}
tree_file <- args[1]
tip_order_file <- args[2]
root <- args[3]
output_pdf <- args[4]
width <- as.numeric(args[5])
height <- as.numeric(args[6])
library(ape)
library(phangorn)
trees <- read.tree(tree_file)
pdf(output_pdf, width = width, height = height)
for(i in 1:length(trees)) {
trees[[i]] <- compute.brlen(root(trees[[i]], root))
}
tip_order <- rev(readLines(tip_order_file))
densiTree(trees, consensus=tip_order, alpha = 0.01,
col = "#009900", type = "cladogram",
label.offset = 0.02, scale.bar = FALSE
)
dev.off()
@@ -0,0 +1,18 @@
#! /usr/bin/env bash
if [ "$#" -ne 3 ]; then
echo "Usage: $0 <input_fasta_dir> <extension> <output_nexus_dir>"
exit 1
fi
input_dir=$1
extension=$2
output_dir=$3
mkdir -p "${output_dir}"
for f in "${input_dir}"/*."${extension}"; do
filename=$(basename -- "${f}")
filename_noext="${filename%.*}"
output_file="${output_dir}/${filename_noext}.nex"
echo "Converting ${f} to ${output_file}"
seqmagick convert --output-format nexus --alphabet dna --input-format fasta "${f}" "${output_file}"
done
+21
View File
@@ -0,0 +1,21 @@
#!/usr/bin/env bash
mkdir -p ../mbsum_out
echo -n "" > ../mbsum.log
for i in *.nex.tar.gz; do
base=$(basename "$i" .nex.tar.gz)
echo "Processing ${base}" >> ../mbsum.log
mkdir -p "${base}"
tar -xzf "$i" -C "${base}"
## skip Average standard deviation of split frequencies > 0.01
dsf=$(awk '/Average standard deviation of split frequencies:/ {out=$7} END{print out+0}' "${base}"/*.nex.log 2>/dev/null)
if awk -v d="$dsf" 'BEGIN{if (d >= 0.01) exit 0; exit 1}'; then
echo "Skipping ${base} due to high DSF: ${dsf}" >> ../mbsum.log
continue
else
mbsum "${base}"/*.t -n 1000 -o ../mbsum_out/"${base}".in >> ../mbsum.log 2>&1
fi
rm -rf "${base}"
echo "Completed ${base}" >> ../mbsum.log
done
echo "All done!"
@@ -0,0 +1,12 @@
begin mrbayes;
set nowarnings=yes;
set usebeagle=yes;
set autoclose=yes;
set seed=12345;
set swapseed=12345;
lset nst=6 rates=gamma;
mcmcp ngen=2000000 burninfrac=.25 samplefreq=1000 printfreq=10000 checkpoint=no
diagnfreq=10000 nruns=3 nchains=3 temp=0.40 swapfreq=10 stoprule=no;
mcmc;
sumt;
end;