Add Vercel configuration for rewrites and build settings

This commit is contained in:
2025-12-17 17:47:45 +08:00
parent 072524e303
commit 0d31d42de4
32 changed files with 7435 additions and 289 deletions
+51
View File
@@ -0,0 +1,51 @@
#! /bin/bash
set -e
if [ "$#" -ne 4 ]; then
echo "Usage: $0 <ogs_dir> <outdir> <proteome> <threads>"
echo "search homologous sequences in <proteome> using HMMs built from orthogroup alignments"
exit 1
fi
ogs_dir=$(readlink -f "$1")
outdir=$2
proteome=$(readlink -f "$3")
threads=$4
mkdir -p "$outdir"
cd "$outdir" || exit 1
echo "Working directory: $(pwd)"
echo "Using OGS directory: $ogs_dir"
echo "Using $threads threads"
echo ""
echo "Starting orthogroup sequence alignment..."
mkdir -p msa
echo -n >mafft.cmds
for i in "$ogs_dir"/*.fa; do
j=$(basename "$i")
echo "linsi --quiet $i > msa/$j" >>mafft.cmds
done
xargs -t -P "$threads" -I cmd -a mafft.cmds bash -c "cmd"
echo "Orthogroup sequence alignment completed."
echo ""
echo "Starting HMM building from alignments..."
mkdir -p hmms
echo -n >hmmbuild.cmds
for i in msa/*.fa; do
j=$(basename "$i")
echo "hmmbuild -o hmms/${j}.hmmbuild.out --amino hmms/${j}.hmm $i" >>hmmbuild.cmds
done
xargs -t -P "$threads" -I cmd -a hmmbuild.cmds bash -c "cmd"
echo "HMM building completed."
echo ""
echo "Starting HMM search against other proteome..."
mkdir -p search
echo -n >hmmsearch.cmds
for i in hmms/*.hmm; do
j=$(basename "$i")
echo "hmmsearch --tblout search/${j}search.tblout $i $proteome > search/${j}search.rawout" >>hmmsearch.cmds
done
xargs -t -P "$threads" -I cmd -a hmmsearch.cmds bash -c "cmd"
echo "HMM search completed."
echo ""
echo "All steps completed successfully."
@@ -1,8 +0,0 @@
#! /usr/bin/env bash
mkdir -p msa
echo -n > mafft.cmds
for i in ogs/*.fa ; do
j=$(basename "$i")
echo "linsi --quiet $i > msa/$j" >> mafft.cmds
done
xargs -t -P 8 -I cmd -a mafft.cmds bash -c "cmd"
@@ -1,8 +0,0 @@
#! /usr/bin/env bash
mkdir -p hmms
echo -n > hmmbuild.cmds
for i in msa/*.fa ; do
j=$(basename "$i")
echo "hmmbuild -o hmms/${j}.hmmbuild.out --amino hmms/${j}.hmm $i" >> hmmbuild.cmds
done
xargs -t -P 8 -I cmd -a hmmbuild.cmds bash -c "cmd"
+34
View File
@@ -0,0 +1,34 @@
#! /bin/bash
set -e
SCRIPTS=${SCRIPTS:-"$PROJECTHOME/99.scripts"}
THREADS=${THREADS:-12}
if [ "$#" -ne 5 ]; then
echo "Usage: $0 <ogs_dir> <hmmsearch_result_dir> <all_cds.fa> <output_dir> <homolog_stem>"
echo "Integrate hmmsearch results to new orthologous groups directory and perform MACSE alignment"
exit 1
fi
ogs_dir=$(readlink -f "$1")
search_dir=$(readlink -f "$2")
all_cds=$(readlink -f "$3")
out_dir=$4
stem=$5
echo "Integrating hmmsearch results to new orthologous groups directory..."
python3 "$SCRIPTS"/miscs/hmmsearch_result_to_new_ogs_dir.py \
-d "$ogs_dir" \
-t "$search_dir" \
-f "$all_cds" \
-o "$out_dir" \
-s "$stem"
echo "Integration completed."
echo "Starting MACSE alignment of orthologous groups..."
echo -n >macse.cmds
for og_dir in "$out_dir"/ogs/*; do
j=$(basename "$og_dir")
echo "cd $og_dir && bash $SCRIPTS/miscs/macse.sh ${j}_${stem}.fa ${j}.fa $j" >>macse.cmds
done
xargs -t -P "$THREADS" -I cmd -a macse.cmds bash -c "cmd"
echo "MACSE alignment completed."
@@ -1,8 +0,0 @@
#! /usr/bin/env bash
mkdir -p hmmsearch
echo -n > hmmsearch.cmds
for i in hmms/*.hmm ; do
j=$(basename "$i")
echo "hmmsearch --tblout hmmsearch/${j}search.tblout $i ../../01.reference/Zju.pep.fa > hmmsearch/${j}search.rawout" >> hmmsearch.cmds
done
xargs -t -P 8 -I cmd -a hmmsearch.cmds bash -c "cmd"
@@ -1,8 +0,0 @@
#! /usr/bin/env bash
mkdir -p pep_aln
echo -n > mafft.cmds
for i in raw_ogs/pep/*.fa; do
j=$(basename "$i")
echo "linsi --quiet $i > pep_aln/${j/.fa/.pal}" >> mafft.cmds
done
xargs -t -P 8 -I cmd -a mafft.cmds bash -c "cmd"
@@ -1,8 +0,0 @@
#! /usr/bin/env bash
mkdir -p cds_aln
echo -n > pal2nal.cmds
for i in pep_aln/*.pal; do
j=$(basename "$i")
echo "pal2nal.pl $i raw_ogs/cds/${j/.pal/.fa} -output fasta > cds_aln/${j/.pal/.nal}" >> pal2nal.cmds
done
xargs -t -P 8 -I cmd -a pal2nal.cmds bash -c "cmd"
@@ -94,11 +94,18 @@ def concatenate_fasta_files(fasta_files, output_file):
print(f"Total output sequence length: {len(concatenated_sequences[0].seq)}.")
def get_fasta_files_from_directory(directory, extensions):
def get_fasta_files_from_directory(directory, extensions, list_file=None):
"""
get all FASTA files from a directory with specified extensions
"""
fasta_files = []
if list_file:
with open(list_file, "r") as lf:
for line in lf:
filepath = os.path.join(directory, line.strip())
if os.path.isfile(filepath):
fasta_files.append(filepath)
return sorted(fasta_files)
for filename in os.listdir(directory):
if any(filename.endswith(ext) for ext in extensions):
fasta_files.append(os.path.join(directory, filename))
@@ -119,12 +126,15 @@ def main():
default=[".fasta", ".fa", ".fna"],
help="FASTA file extensions to look for in directory",
)
parser.add_argument("-l", "--list", help="List of input FASTA files", default=None)
args = parser.parse_args()
# 获取输入文件
if args.directory:
fasta_files = get_fasta_files_from_directory(args.directory, args.extensions)
fasta_files = get_fasta_files_from_directory(
args.directory, args.extensions, args.list
)
if not fasta_files:
print(
f"Cannot find FASTA files in {args.directory} with extensions {args.extensions}"
@@ -137,7 +147,7 @@ def main():
return
print(f"Found {len(fasta_files)} FASTA files:")
# Perform concatenation
concatenate_fasta_files(fasta_files, args.output)
+15
View File
@@ -0,0 +1,15 @@
#! /usr/bash
if [ "$#" -ne 5 ]; then
echo "Usage: $0 <reference_genome> <fastq_1> <fastq_2> <output_directory> <stem>"
exit 1
fi
ref=$1
fq1=$2
fq2=$3
outdir=$4
stem=$5
mkdir -p "$outdir"
hisat -p 4 --dta -x "$ref" -1 "$fq1" -2 "$fq2" -S "${outdir}/${stem}.sam"
samtools view -b -@ 4 "${outdir}/${stem}.sam" | samtools sort -@ 4 -o "${outdir}/${stem}.sorted.bam" --write-index
rm "${outdir}/${stem}.sam"
echo "Mapping completed. Sorted BAM file is at ${outdir}/${stem}.sorted.bam"
@@ -0,0 +1,31 @@
#! /bin/bash
set -e
SCRIPTS=${SCRIPTS:-"$PROJECTHOME/99.scripts"}
MAX_MEMORY=${MAX_MEMORY:-"20G"}
VIRIDI=${VIRIDI:-"$PROJECTHOME/01.reference/viridiplantae_odb12/"}
ROSALES=${ROSALES:-"$PROJECTHOME/01.reference/rosales_odb12/"}
if [ "$#" -ne 4 ]; then
echo "Usage: $0 <reads_1.fastq> <reads_2.fastq> <stem> <threads>"
echo "Perform de novo transcriptome assembly using Trinity"
exit 1
fi
fq1=$1
fq2=$2
stem=$3
outdir="$stem"_trinity_out_dir
THREADS=$4
# Run Trinity for de novo transcriptome assembly
Trinity --seqType fq --left "$fq1" --right "$fq2" --CPU "$THREADS" --max_memory "$MAX_MEMORY" --output "$outdir"
# Get Longest isoform per gene
perl "$SCRIPTS"/trinity_utils/util/misc/get_longest_isoform_seq_per_trinity_gene.pl "$outdir.Trinity.fasta" >"$outdir".longest_isoform.fasta
# BUSCO assessment
busco -i "$outdir".longest_isoform.fasta -l "$VIRIDI" -m tran --cpu "$THREADS" -o "$outdir"_busco_viridi -f
busco -i "$outdir".longest_isoform.fasta -l "$ROSALES" -m tran --cpu "$THREADS" -o "$outdir"_busco_rosales -f
# Length Statistics
TrinityStats.pl "$outdir.Trinity.fasta" >"$outdir".Trinity.fasta.length_stat.txt
# Clear temporary directory
rm -rf "$outdir"
@@ -0,0 +1,37 @@
#! /bin/bash
set -e
MAX_MEMORY=${MAX_MEMORY:-"50G"}
VIRIDI=${VIRIDI:-"$PROJECTHOME/01.reference/viridiplantae_odb12/"}
ROSALES=${ROSALES:-"$PROJECTHOME/01.reference/rosales_odb12/"}
SCRIPTS=${SCRIPTS:-"$PROJECTHOME/99.scripts"}
if [ "$#" -ne 5 ]; then
echo "Usage: $0 <reads_1.fastq> <reads_2.fastq> <ref> <stem> <threads>"
echo "Perform reference-guided transcriptome assembly using Hisat2 and Trinity"
exit 1
fi
fq1=$1
fq2=$2
ref=$3
stem=$4
outdir="$stem"_trinity_out_dir
THREADS=$5
# Run Hisat2 for reads mapping to reference genome
hisat2 -p "$THREADS" --dta -x "$ref" -1 "$fq1" -2 "$fq2" -S "$stem".sam
samtools view -b -@ "$THREADS" -o "$stem".raw.bam "$stem".sam
samtools sort -@ "$THREADS" -o "$stem".sorted.bam "$stem".raw.bam
samtools index "$stem".sorted.bam
rm "$stem".sam "$stem".raw.bam
# Run Trinity for de novo transcriptome assembly
Trinity --genome_guided_bam "$stem".sorted.bam --genome_guided_max_intron 10000 --max_memory "$MAX_MEMORY" --CPU "$THREADS" --output "$outdir"
# Get Longest isoform per gene
perl "$SCRIPTS"/trinity_utils/util/misc/get_longest_isoform_seq_per_trinity_gene.pl "$outdir/Trinity-GG.fasta" >"$outdir/longest_isoform.fasta"
# BUSCO assessment
busco -i "$outdir"/longest_isoform.fasta -l "$VIRIDI" -m tran --cpu "$THREADS" -o "$outdir"/busco_viridi -f
busco -i "$outdir"/longest_isoform.fasta -l "$ROSALES" -m tran --cpu "$THREADS" -o "$outdir"/busco_rosales -f
# Length Statistics
TrinityStats.pl "$outdir"/Trinity-GG.fasta >"$outdir"/length_stat.txt
@@ -0,0 +1,18 @@
#! /bin/bash
set -e
TMP=${TMP:-"$PROJECTHOME/tmp"}
if [ "$#" -ne 3 ]; then
echo "Usage: $0 <transcripts_fasta> <swissprot_database> <output_directory>"
echo "Predict coding sequences (CDS) from transcripts using TD2 and MMseqs2"
exit 1
fi
transcripts=$1
sprot=$2
outdir=$3
mkdir -p "$outdir"
TD2.LongOrfs -t "$transcripts" --precise -@ 8 -O "$outdir"
mmseqs easy-search "$outdir/longest_orfs.pep" "$sprot" "$outdir/mmseqs.m8" "$TMP" -s 7.0 --threads 16
TD2.Predict -t "$transcripts" --precise -O "$outdir" --retain-mmseqs-hits "$outdir/mmseqs.m8"
echo "CDS prediction completed. Results are in $outdir"
@@ -0,0 +1,25 @@
#! /bin/bash
set -e
SCRIPTS=${SCRIPTS:-"$PROJECTHOME/99.scripts"}
if [ "$#" -ne 3 ]; then
echo "Usage: $0 <input_dir> <output_dir> <extension>"
echo "Extract longest isoform per gene and rename sequences"
exit 1
fi
indir=$1
outdir=$2
ext=$3
mkdir -p "$outdir"
# Process each file in the input directory with the specified extension
for td_cds in "$indir"/*."$ext"; do
stem=$(basename "$td_cds" ."$ext")
echo "Processing $td_cds($stem) ..."
echo "perl $SCRIPTS/trinity_utils/util/misc/get_longest_isoform_seq_per_trinity_gene.pl $td_cds > $outdir/$stem.longest_isoform.fa"
perl "$SCRIPTS"/trinity_utils/util/misc/get_longest_isoform_seq_per_trinity_gene.pl "$td_cds" >"$outdir"/"$stem".longest_isoform.fa
echo "$SCRIPTS/rename_trinity_fasta.py $outdir/$stem.longest_isoform.fa $stem $outdir/$stem.full_cds.fa"
"$SCRIPTS"/miscs/rename_trinity_fasta.py "$outdir"/"$stem".longest_isoform.fa "$stem" "$outdir"/"$stem".full_cds.fa
echo "Done."
done
@@ -0,0 +1,27 @@
#! /bin/bash
set -e
SCRIPTS=${SCRIPTS:-"$PROJECTHOME/99.scripts"}
IDENTITY=${IDENTITY:-0.99}
THREADS=${THREADS:-6}S
if [ "$#" -ne 3 ]; then
echo "Usage: $0 <input_dir> <output_dir> <extension>"
echo "Reduce redundancy of CDS files using cd-hit-est"
exit 1
fi
indir=$1
outdir=$2
ext=$3
mkdir -p "$outdir"
# Process each file in the input directory with the specified extension
for cds in "$indir"/*."$ext"; do
stem=$(basename "$cds" ."$ext")
echo "Processing $cds($stem) ..."
echo "cd-hit-est -i $cds -o $outdir/$stem.cds_rr.fa -c $IDENTITY -n 10 -r 0 -T $THREADS"
cd-hit-est -i "$cds" -o "$outdir/$stem".cds_rr.fa -c "$IDENTITY" -n 10 -r 0 -T "$THREADS"
echo "seqkit translate $outdir/$stem.cds_rr.fa > $outdir/$stem.prot_rr.fa"
seqkit translate "$outdir/$stem".cds_rr.fa >"$outdir/$stem".prot_rr.fa
echo "Done."
done