Add Vercel configuration for rewrites and build settings
This commit is contained in:
+51
@@ -0,0 +1,51 @@
|
||||
#! /bin/bash
|
||||
set -e
|
||||
|
||||
if [ "$#" -ne 4 ]; then
|
||||
echo "Usage: $0 <ogs_dir> <outdir> <proteome> <threads>"
|
||||
echo "search homologous sequences in <proteome> using HMMs built from orthogroup alignments"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
ogs_dir=$(readlink -f "$1")
|
||||
outdir=$2
|
||||
proteome=$(readlink -f "$3")
|
||||
threads=$4
|
||||
|
||||
mkdir -p "$outdir"
|
||||
cd "$outdir" || exit 1
|
||||
echo "Working directory: $(pwd)"
|
||||
echo "Using OGS directory: $ogs_dir"
|
||||
echo "Using $threads threads"
|
||||
echo ""
|
||||
echo "Starting orthogroup sequence alignment..."
|
||||
mkdir -p msa
|
||||
echo -n >mafft.cmds
|
||||
for i in "$ogs_dir"/*.fa; do
|
||||
j=$(basename "$i")
|
||||
echo "linsi --quiet $i > msa/$j" >>mafft.cmds
|
||||
done
|
||||
xargs -t -P "$threads" -I cmd -a mafft.cmds bash -c "cmd"
|
||||
echo "Orthogroup sequence alignment completed."
|
||||
echo ""
|
||||
echo "Starting HMM building from alignments..."
|
||||
mkdir -p hmms
|
||||
echo -n >hmmbuild.cmds
|
||||
for i in msa/*.fa; do
|
||||
j=$(basename "$i")
|
||||
echo "hmmbuild -o hmms/${j}.hmmbuild.out --amino hmms/${j}.hmm $i" >>hmmbuild.cmds
|
||||
done
|
||||
xargs -t -P "$threads" -I cmd -a hmmbuild.cmds bash -c "cmd"
|
||||
echo "HMM building completed."
|
||||
echo ""
|
||||
echo "Starting HMM search against other proteome..."
|
||||
mkdir -p search
|
||||
echo -n >hmmsearch.cmds
|
||||
for i in hmms/*.hmm; do
|
||||
j=$(basename "$i")
|
||||
echo "hmmsearch --tblout search/${j}search.tblout $i $proteome > search/${j}search.rawout" >>hmmsearch.cmds
|
||||
done
|
||||
xargs -t -P "$threads" -I cmd -a hmmsearch.cmds bash -c "cmd"
|
||||
echo "HMM search completed."
|
||||
echo ""
|
||||
echo "All steps completed successfully."
|
||||
@@ -1,8 +0,0 @@
|
||||
#! /usr/bin/env bash
|
||||
mkdir -p msa
|
||||
echo -n > mafft.cmds
|
||||
for i in ogs/*.fa ; do
|
||||
j=$(basename "$i")
|
||||
echo "linsi --quiet $i > msa/$j" >> mafft.cmds
|
||||
done
|
||||
xargs -t -P 8 -I cmd -a mafft.cmds bash -c "cmd"
|
||||
@@ -1,8 +0,0 @@
|
||||
#! /usr/bin/env bash
|
||||
mkdir -p hmms
|
||||
echo -n > hmmbuild.cmds
|
||||
for i in msa/*.fa ; do
|
||||
j=$(basename "$i")
|
||||
echo "hmmbuild -o hmms/${j}.hmmbuild.out --amino hmms/${j}.hmm $i" >> hmmbuild.cmds
|
||||
done
|
||||
xargs -t -P 8 -I cmd -a hmmbuild.cmds bash -c "cmd"
|
||||
@@ -0,0 +1,34 @@
|
||||
#! /bin/bash
|
||||
set -e
|
||||
SCRIPTS=${SCRIPTS:-"$PROJECTHOME/99.scripts"}
|
||||
THREADS=${THREADS:-12}
|
||||
|
||||
if [ "$#" -ne 5 ]; then
|
||||
echo "Usage: $0 <ogs_dir> <hmmsearch_result_dir> <all_cds.fa> <output_dir> <homolog_stem>"
|
||||
echo "Integrate hmmsearch results to new orthologous groups directory and perform MACSE alignment"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
ogs_dir=$(readlink -f "$1")
|
||||
search_dir=$(readlink -f "$2")
|
||||
all_cds=$(readlink -f "$3")
|
||||
out_dir=$4
|
||||
stem=$5
|
||||
|
||||
echo "Integrating hmmsearch results to new orthologous groups directory..."
|
||||
python3 "$SCRIPTS"/miscs/hmmsearch_result_to_new_ogs_dir.py \
|
||||
-d "$ogs_dir" \
|
||||
-t "$search_dir" \
|
||||
-f "$all_cds" \
|
||||
-o "$out_dir" \
|
||||
-s "$stem"
|
||||
echo "Integration completed."
|
||||
|
||||
echo "Starting MACSE alignment of orthologous groups..."
|
||||
echo -n >macse.cmds
|
||||
for og_dir in "$out_dir"/ogs/*; do
|
||||
j=$(basename "$og_dir")
|
||||
echo "cd $og_dir && bash $SCRIPTS/miscs/macse.sh ${j}_${stem}.fa ${j}.fa $j" >>macse.cmds
|
||||
done
|
||||
xargs -t -P "$THREADS" -I cmd -a macse.cmds bash -c "cmd"
|
||||
echo "MACSE alignment completed."
|
||||
@@ -1,8 +0,0 @@
|
||||
#! /usr/bin/env bash
|
||||
mkdir -p hmmsearch
|
||||
echo -n > hmmsearch.cmds
|
||||
for i in hmms/*.hmm ; do
|
||||
j=$(basename "$i")
|
||||
echo "hmmsearch --tblout hmmsearch/${j}search.tblout $i ../../01.reference/Zju.pep.fa > hmmsearch/${j}search.rawout" >> hmmsearch.cmds
|
||||
done
|
||||
xargs -t -P 8 -I cmd -a hmmsearch.cmds bash -c "cmd"
|
||||
@@ -1,8 +0,0 @@
|
||||
#! /usr/bin/env bash
|
||||
mkdir -p pep_aln
|
||||
echo -n > mafft.cmds
|
||||
for i in raw_ogs/pep/*.fa; do
|
||||
j=$(basename "$i")
|
||||
echo "linsi --quiet $i > pep_aln/${j/.fa/.pal}" >> mafft.cmds
|
||||
done
|
||||
xargs -t -P 8 -I cmd -a mafft.cmds bash -c "cmd"
|
||||
@@ -1,8 +0,0 @@
|
||||
#! /usr/bin/env bash
|
||||
mkdir -p cds_aln
|
||||
echo -n > pal2nal.cmds
|
||||
for i in pep_aln/*.pal; do
|
||||
j=$(basename "$i")
|
||||
echo "pal2nal.pl $i raw_ogs/cds/${j/.pal/.fa} -output fasta > cds_aln/${j/.pal/.nal}" >> pal2nal.cmds
|
||||
done
|
||||
xargs -t -P 8 -I cmd -a pal2nal.cmds bash -c "cmd"
|
||||
@@ -94,11 +94,18 @@ def concatenate_fasta_files(fasta_files, output_file):
|
||||
print(f"Total output sequence length: {len(concatenated_sequences[0].seq)}.")
|
||||
|
||||
|
||||
def get_fasta_files_from_directory(directory, extensions):
|
||||
def get_fasta_files_from_directory(directory, extensions, list_file=None):
|
||||
"""
|
||||
get all FASTA files from a directory with specified extensions
|
||||
"""
|
||||
fasta_files = []
|
||||
if list_file:
|
||||
with open(list_file, "r") as lf:
|
||||
for line in lf:
|
||||
filepath = os.path.join(directory, line.strip())
|
||||
if os.path.isfile(filepath):
|
||||
fasta_files.append(filepath)
|
||||
return sorted(fasta_files)
|
||||
for filename in os.listdir(directory):
|
||||
if any(filename.endswith(ext) for ext in extensions):
|
||||
fasta_files.append(os.path.join(directory, filename))
|
||||
@@ -119,12 +126,15 @@ def main():
|
||||
default=[".fasta", ".fa", ".fna"],
|
||||
help="FASTA file extensions to look for in directory",
|
||||
)
|
||||
parser.add_argument("-l", "--list", help="List of input FASTA files", default=None)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# 获取输入文件
|
||||
if args.directory:
|
||||
fasta_files = get_fasta_files_from_directory(args.directory, args.extensions)
|
||||
fasta_files = get_fasta_files_from_directory(
|
||||
args.directory, args.extensions, args.list
|
||||
)
|
||||
if not fasta_files:
|
||||
print(
|
||||
f"Cannot find FASTA files in {args.directory} with extensions {args.extensions}"
|
||||
@@ -137,7 +147,7 @@ def main():
|
||||
return
|
||||
|
||||
print(f"Found {len(fasta_files)} FASTA files:")
|
||||
|
||||
|
||||
# Perform concatenation
|
||||
concatenate_fasta_files(fasta_files, args.output)
|
||||
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
#! /usr/bash
|
||||
if [ "$#" -ne 5 ]; then
|
||||
echo "Usage: $0 <reference_genome> <fastq_1> <fastq_2> <output_directory> <stem>"
|
||||
exit 1
|
||||
fi
|
||||
ref=$1
|
||||
fq1=$2
|
||||
fq2=$3
|
||||
outdir=$4
|
||||
stem=$5
|
||||
mkdir -p "$outdir"
|
||||
hisat -p 4 --dta -x "$ref" -1 "$fq1" -2 "$fq2" -S "${outdir}/${stem}.sam"
|
||||
samtools view -b -@ 4 "${outdir}/${stem}.sam" | samtools sort -@ 4 -o "${outdir}/${stem}.sorted.bam" --write-index
|
||||
rm "${outdir}/${stem}.sam"
|
||||
echo "Mapping completed. Sorted BAM file is at ${outdir}/${stem}.sorted.bam"
|
||||
@@ -0,0 +1,31 @@
|
||||
#! /bin/bash
|
||||
set -e
|
||||
SCRIPTS=${SCRIPTS:-"$PROJECTHOME/99.scripts"}
|
||||
MAX_MEMORY=${MAX_MEMORY:-"20G"}
|
||||
VIRIDI=${VIRIDI:-"$PROJECTHOME/01.reference/viridiplantae_odb12/"}
|
||||
ROSALES=${ROSALES:-"$PROJECTHOME/01.reference/rosales_odb12/"}
|
||||
|
||||
if [ "$#" -ne 4 ]; then
|
||||
echo "Usage: $0 <reads_1.fastq> <reads_2.fastq> <stem> <threads>"
|
||||
echo "Perform de novo transcriptome assembly using Trinity"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
fq1=$1
|
||||
fq2=$2
|
||||
stem=$3
|
||||
outdir="$stem"_trinity_out_dir
|
||||
THREADS=$4
|
||||
|
||||
|
||||
# Run Trinity for de novo transcriptome assembly
|
||||
Trinity --seqType fq --left "$fq1" --right "$fq2" --CPU "$THREADS" --max_memory "$MAX_MEMORY" --output "$outdir"
|
||||
# Get Longest isoform per gene
|
||||
perl "$SCRIPTS"/trinity_utils/util/misc/get_longest_isoform_seq_per_trinity_gene.pl "$outdir.Trinity.fasta" >"$outdir".longest_isoform.fasta
|
||||
# BUSCO assessment
|
||||
busco -i "$outdir".longest_isoform.fasta -l "$VIRIDI" -m tran --cpu "$THREADS" -o "$outdir"_busco_viridi -f
|
||||
busco -i "$outdir".longest_isoform.fasta -l "$ROSALES" -m tran --cpu "$THREADS" -o "$outdir"_busco_rosales -f
|
||||
# Length Statistics
|
||||
TrinityStats.pl "$outdir.Trinity.fasta" >"$outdir".Trinity.fasta.length_stat.txt
|
||||
# Clear temporary directory
|
||||
rm -rf "$outdir"
|
||||
@@ -0,0 +1,37 @@
|
||||
#! /bin/bash
|
||||
set -e
|
||||
MAX_MEMORY=${MAX_MEMORY:-"50G"}
|
||||
VIRIDI=${VIRIDI:-"$PROJECTHOME/01.reference/viridiplantae_odb12/"}
|
||||
ROSALES=${ROSALES:-"$PROJECTHOME/01.reference/rosales_odb12/"}
|
||||
SCRIPTS=${SCRIPTS:-"$PROJECTHOME/99.scripts"}
|
||||
|
||||
if [ "$#" -ne 5 ]; then
|
||||
echo "Usage: $0 <reads_1.fastq> <reads_2.fastq> <ref> <stem> <threads>"
|
||||
echo "Perform reference-guided transcriptome assembly using Hisat2 and Trinity"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
fq1=$1
|
||||
fq2=$2
|
||||
ref=$3
|
||||
stem=$4
|
||||
outdir="$stem"_trinity_out_dir
|
||||
THREADS=$5
|
||||
|
||||
|
||||
# Run Hisat2 for reads mapping to reference genome
|
||||
hisat2 -p "$THREADS" --dta -x "$ref" -1 "$fq1" -2 "$fq2" -S "$stem".sam
|
||||
samtools view -b -@ "$THREADS" -o "$stem".raw.bam "$stem".sam
|
||||
samtools sort -@ "$THREADS" -o "$stem".sorted.bam "$stem".raw.bam
|
||||
samtools index "$stem".sorted.bam
|
||||
rm "$stem".sam "$stem".raw.bam
|
||||
# Run Trinity for de novo transcriptome assembly
|
||||
Trinity --genome_guided_bam "$stem".sorted.bam --genome_guided_max_intron 10000 --max_memory "$MAX_MEMORY" --CPU "$THREADS" --output "$outdir"
|
||||
# Get Longest isoform per gene
|
||||
perl "$SCRIPTS"/trinity_utils/util/misc/get_longest_isoform_seq_per_trinity_gene.pl "$outdir/Trinity-GG.fasta" >"$outdir/longest_isoform.fasta"
|
||||
# BUSCO assessment
|
||||
busco -i "$outdir"/longest_isoform.fasta -l "$VIRIDI" -m tran --cpu "$THREADS" -o "$outdir"/busco_viridi -f
|
||||
busco -i "$outdir"/longest_isoform.fasta -l "$ROSALES" -m tran --cpu "$THREADS" -o "$outdir"/busco_rosales -f
|
||||
# Length Statistics
|
||||
TrinityStats.pl "$outdir"/Trinity-GG.fasta >"$outdir"/length_stat.txt
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
#! /bin/bash
|
||||
set -e
|
||||
TMP=${TMP:-"$PROJECTHOME/tmp"}
|
||||
|
||||
if [ "$#" -ne 3 ]; then
|
||||
echo "Usage: $0 <transcripts_fasta> <swissprot_database> <output_directory>"
|
||||
echo "Predict coding sequences (CDS) from transcripts using TD2 and MMseqs2"
|
||||
exit 1
|
||||
fi
|
||||
transcripts=$1
|
||||
sprot=$2
|
||||
outdir=$3
|
||||
|
||||
mkdir -p "$outdir"
|
||||
TD2.LongOrfs -t "$transcripts" --precise -@ 8 -O "$outdir"
|
||||
mmseqs easy-search "$outdir/longest_orfs.pep" "$sprot" "$outdir/mmseqs.m8" "$TMP" -s 7.0 --threads 16
|
||||
TD2.Predict -t "$transcripts" --precise -O "$outdir" --retain-mmseqs-hits "$outdir/mmseqs.m8"
|
||||
echo "CDS prediction completed. Results are in $outdir"
|
||||
@@ -0,0 +1,25 @@
|
||||
#! /bin/bash
|
||||
set -e
|
||||
SCRIPTS=${SCRIPTS:-"$PROJECTHOME/99.scripts"}
|
||||
|
||||
if [ "$#" -ne 3 ]; then
|
||||
echo "Usage: $0 <input_dir> <output_dir> <extension>"
|
||||
echo "Extract longest isoform per gene and rename sequences"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
indir=$1
|
||||
outdir=$2
|
||||
ext=$3
|
||||
|
||||
mkdir -p "$outdir"
|
||||
# Process each file in the input directory with the specified extension
|
||||
for td_cds in "$indir"/*."$ext"; do
|
||||
stem=$(basename "$td_cds" ."$ext")
|
||||
echo "Processing $td_cds($stem) ..."
|
||||
echo "perl $SCRIPTS/trinity_utils/util/misc/get_longest_isoform_seq_per_trinity_gene.pl $td_cds > $outdir/$stem.longest_isoform.fa"
|
||||
perl "$SCRIPTS"/trinity_utils/util/misc/get_longest_isoform_seq_per_trinity_gene.pl "$td_cds" >"$outdir"/"$stem".longest_isoform.fa
|
||||
echo "$SCRIPTS/rename_trinity_fasta.py $outdir/$stem.longest_isoform.fa $stem $outdir/$stem.full_cds.fa"
|
||||
"$SCRIPTS"/miscs/rename_trinity_fasta.py "$outdir"/"$stem".longest_isoform.fa "$stem" "$outdir"/"$stem".full_cds.fa
|
||||
echo "Done."
|
||||
done
|
||||
@@ -0,0 +1,27 @@
|
||||
#! /bin/bash
|
||||
set -e
|
||||
SCRIPTS=${SCRIPTS:-"$PROJECTHOME/99.scripts"}
|
||||
IDENTITY=${IDENTITY:-0.99}
|
||||
THREADS=${THREADS:-6}S
|
||||
|
||||
if [ "$#" -ne 3 ]; then
|
||||
echo "Usage: $0 <input_dir> <output_dir> <extension>"
|
||||
echo "Reduce redundancy of CDS files using cd-hit-est"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
indir=$1
|
||||
outdir=$2
|
||||
ext=$3
|
||||
|
||||
mkdir -p "$outdir"
|
||||
# Process each file in the input directory with the specified extension
|
||||
for cds in "$indir"/*."$ext"; do
|
||||
stem=$(basename "$cds" ."$ext")
|
||||
echo "Processing $cds($stem) ..."
|
||||
echo "cd-hit-est -i $cds -o $outdir/$stem.cds_rr.fa -c $IDENTITY -n 10 -r 0 -T $THREADS"
|
||||
cd-hit-est -i "$cds" -o "$outdir/$stem".cds_rr.fa -c "$IDENTITY" -n 10 -r 0 -T "$THREADS"
|
||||
echo "seqkit translate $outdir/$stem.cds_rr.fa > $outdir/$stem.prot_rr.fa"
|
||||
seqkit translate "$outdir/$stem".cds_rr.fa >"$outdir/$stem".prot_rr.fa
|
||||
echo "Done."
|
||||
done
|
||||
Reference in New Issue
Block a user