Add scripts for phylogenetic analysis using MrBayes and RAxML
- Created `mbblock.txt` for MrBayes configuration. - Implemented `mrbayes.sh` to convert FASTA to Nexus and run MrBayes. - Developed `raxml.sh` to run modeltest-ng and raxml-ng on alignments. - Added `04.filter_ogs.sh` to filter orthologous groups based on taxon count and sequence length. - Implemented `01.run_mrbayes.sh` to execute MrBayes for multiple FASTA files. - Created `02.mbsum.sh` to summarize MrBayes output while skipping high DSF analyses. - Developed `03.run_raxml.sh` to run raxml-ng on filtered alignments.
This commit is contained in:
@@ -0,0 +1,23 @@
|
||||
#! /bin/bash
|
||||
set -e
|
||||
if [ "$#" -ne 4 ]; then
|
||||
echo "Usage: $0 <out_dir> <treeshrink_dir> <min_taxon> <min_seq_length>"
|
||||
echo "Filter orthologous groups after TreeShrink based on minimum taxon number and minimum sequence length"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
out_dir=$(readlink -f "$1")
|
||||
treeshrink_dir=$(readlink -f "$2")
|
||||
min_taxon=$3
|
||||
min_seq_length=$4
|
||||
|
||||
mkdir -p "$out_dir"
|
||||
for i in "$treeshrink_dir"/*/output.fasta; do
|
||||
j=$(dirname "$i")
|
||||
j=$(basename "$j")
|
||||
seqlen=$(seqkit fx2tab -C ATCG "$i" | awk '{print $3}' | sort -n | head -n 1)
|
||||
seqnum=$(grep -c ">" "$i")
|
||||
if [[ $seqnum -eq $min_taxon && $seqlen -ge $min_seq_length ]]; then
|
||||
cp -l "$i" "$out_dir"/"${j}.fa"
|
||||
fi
|
||||
done
|
||||
@@ -1,12 +0,0 @@
|
||||
#! /usr/bin/env bash
|
||||
total_taxon=11
|
||||
min_seq_length=300
|
||||
mkdir -p final_ogs
|
||||
for i in treeshrink/* ; do
|
||||
j=$(basename "$i")
|
||||
seqlen=$(seqkit fx2tab -C ATCG "$i"/output.fasta | awk '{print $3}' | sort -n | head -n 1)
|
||||
seqnum=$(grep -c ">" "$i"/output.fasta)
|
||||
if [[ $seqnum -eq $total_taxon && $seqlen -ge $min_seq_length ]]; then
|
||||
cp -l "$i"/output.fasta final_ogs/"${j}.fa"
|
||||
fi
|
||||
done
|
||||
Reference in New Issue
Block a user