This commit is contained in:
2025-11-25 00:28:51 +08:00
commit eb3f16c30e
406 changed files with 91653 additions and 0 deletions
@@ -0,0 +1,291 @@
#!/usr/bin/env perl
=pod
=head1 NAME
=head1 USAGE
-in input file in FASTA/Q or posmap length file (e.g. posmap.scflen)
-genome genome size in bp for estimating N lengths and indexes
-single FASTA/Q has sequence in a single line (faster)
-overwrite => Force overwrite
-reads => Force processing as read data (no N50 statistics)
-noreads => Force as not being read data. Good for cDNA assemblies with short contigs
=head1 AUTHORS
Alexie Papanicolaou 1
Ecosystem Sciences, CSIRO, Black Mountain Labs, Clunies Ross Str, Canberra, Australia
alexie@butterflybase.org
=head1 DISCLAIMER & LICENSE
This software is released under the GNU General Public License version 3 (GPLv3).
It is provided "as is" without warranty of any kind.
You can find the terms and conditions at http://www.opensource.org/licenses/gpl-3.0.html.
Please note that incorporating the whole software or parts of its code in proprietary software
is prohibited under the current license.
=head1 BUGS & LIMITATIONS
None known so far.
=cut
use strict;
use warnings;
use Getopt::Long;
use Pod::Usage;
use Statistics::Descriptive;
use Bio::SeqIO;
$|=1;
my (@infiles,$user_genome_size,$is_fasta,$is_fastq,$is_single,$overwrite,$is_reads,$isnot_reads);
GetOptions(
'in=s{,}' => \@infiles,
'single' =>\$is_single,
'genome:s' => \$user_genome_size,
'overwrite' => \$overwrite,
'reads' =>\$is_reads,
'noreads' =>\$isnot_reads,
);
if (!@infiles){
@infiles = @ARGV;
}
pod2usage "No input files!\n" if !@infiles;
die "Cannot ask for both reads and noreads options at the same time!\n" if $is_reads && $isnot_reads;
if ($is_reads && !$isnot_reads){
print "Processing all data as reads\n";
}
if ($user_genome_size && $user_genome_size=~/\D/){
if ($user_genome_size=~/^(\d+)kb$/i){
$user_genome_size=int($1.'000');
}
elsif ($user_genome_size=~/^(\d+)mb$/i){
$user_genome_size=int($1.'000000');
}
elsif ($user_genome_size=~/^(\d+)gb$/i){
$user_genome_size=int($1.'000000000');
}
print "Genome set to ".&thousands($user_genome_size)." b.p.\n";
}
foreach my $infile (@infiles){
unless ($infile && -s $infile){warn("I need a posmap length file, e.g. .posmap.scflen for scaffolds\n");pod2usage;}
my $outfile=$infile.'.n50';
$outfile.='g' if ($user_genome_size);
warn ("Outfile $outfile already exists\n") if -s $outfile && !$overwrite;
next if -s $outfile && !$overwrite;
my $total=int(0);
my $gaps = int(0);
my $seq_ref;
my @head=`head $infile`;
foreach (@head){
if ($_=~/^>\S/){
$is_fasta=1;
print "FASTA file found!\n";
last;
}elsif($_=~/^@\S/){
$is_fastq=1;
print "FASTQ file found!\n";
last;
}
}
print "Parsing file $infile...\n";
if ($is_fasta){
($total,$gaps,$seq_ref) = &process_fasta($infile);
}
elsif($is_fastq){
($total,$gaps,$seq_ref) = &process_fastq($infile);
}
else {
($total,$gaps,$seq_ref) = &process_csv($infile);
}
print "Preparing stats...\n";
my ($mean,$n50,$n10,$n25,$n50_length,$n10_length,$n25_length,$scaffolds,$scaffolds_size,$smallest,$largest,$sequence_number,$sum,$genome_size) = &process_stats($seq_ref,$total);
if ($mean){
open (OUT,">".$outfile);
my $stat = Statistics::Descriptive::Full->new();
$stat->add_data($seq_ref);
#my $skew='';sprintf("%.2f",$stat->skewness());
my $mean = sprintf("%.2f",$mean);
my $median = $stat->median();
my $var = sprintf("%.2f",$stat->variance());
my $sd = sprintf("%.2f",$stat->standard_deviation());
if (!$scaffolds || $scaffolds == 0){
$scaffolds=$sequence_number;
$scaffolds_size=$smallest;
}
print OUT "File: $infile\n";
print OUT "TOTAL: ".&thousands($total)." bp in ".&thousands($sequence_number)." sequences\n";
print OUT "\tof which ".&thousands($gaps)." are Ns/gaps.\n";
print OUT "Mean: ".&thousands($mean)."\nStdev: ".&thousands($sd)."\n";
print OUT "Median: ".&thousands($median)."\n";
print OUT "Smallest: ".&thousands($smallest)."\nLargest: ".&thousands($largest)."\n";
if (($mean >=1000 && !$is_reads) || $isnot_reads){
print OUT "N10 length: ".&thousands($n10_length)."\nN10 Number: ".&thousands($n10)."\n";
print OUT "N25 length: ".&thousands($n25_length)."\nN25 Number: ".&thousands($n25)."\n";
print OUT "N50 length: ".&thousands($n50_length)."\nN50 Number: ".&thousands($n50)."\n";
print OUT "Assuming a genome size of "
.&thousands($user_genome_size)
." then the top ".&thousands($scaffolds)
." account for it (min "
.&thousands($scaffolds_size)
." bp)\n" if $user_genome_size;
}else{
print OUT "Reads found! Read coverage estimated to ".sprintf("%.2f",$total/$user_genome_size)."x using user provided genome size of ".&thousands($user_genome_size)."\n" if $user_genome_size;
}
close (OUT);
print "Done, see $outfile\n";
system("cat $outfile");
}else {
open (OUT,">".$outfile);
print OUT "File: $infile\n";
print OUT "TOTAL: $total bp in $sequence_number sequences\n";
close (OUT);
warn "Non fatal warning: Something went wrong in estimating the statistics. Maybe the provided genome length is much larger than sequence length or maybe less than 3 sequences provided?\n";
}
}
########################################################################
sub process_fasta(){
print "Processing as FASTA\n";
my $infile=shift;
my @array ;
my $total=int(0);
my $gaps=int(0);
my $counter = int(0);
if ($is_single){
open (IN,$infile)||die;
while (my $seq_id=<IN>) {
my $seq=<IN>;
my $length=length($seq)-1; # newline
$counter+=length($seq_id)+$length+1;
next unless $length;
$gaps+=($seq=~tr/[N\-]//);
push(@array,$length);
$total+=$length;
}
close IN;
}else{
my $filein = new Bio::SeqIO(-file=>$infile , -format=>'fasta');
while (my $seq_obj=$filein->next_seq()) {
$counter+=length($seq_obj->seq().$seq_obj->description().' '.$seq_obj->id()) if $seq_obj->seq();
my $length=$seq_obj->length();
my $seq=$seq_obj->seq();
$gaps+=($seq=~tr/[N\-]//);
next unless $length;
push(@array,$length);
$total+=$length;
}
}
print "\n";
die "No data found or wrong format\n" unless $total;
return ($total,$gaps,\@array);
}
sub process_fastq(){
print "Processing as FASTQ\n";
my $infile=shift;
my @array ;
my $total=int(0);
my $gaps=int(0);
my $counter = int(0);
open (IN,$infile);
while (my $seq_id=<IN>) {
my $seq=<IN>;
my $scrap=<IN>.<IN>;
my $length=length($seq)-1; #newline
$counter+=length($seq_id)+$length+1;
next unless $length;
$gaps+=($seq=~tr/[N\-]//);
push(@array,$length);
$total+=$length;
}
close IN;
print "\n";
die "No data found or wrong format\n" unless $total;
return ($total,$gaps,\@array);
}
sub process_csv(){
print "Processing as CSV\n";
my $infile=shift;
my @array ;
my $total=int(0);
my $gaps='N/A';
my $counter = int(0);
open (IN,$infile)||die ("Cannot open $infile\n");
while (my $ln=<IN>){
$counter+=length($ln);
$ln=~/(\d+)$/;
next unless $1;
my $length= $1;
push(@array,$1);
$total+=$length;
}
close IN;
die "No data found or wrong format\n" unless $total;
return ($total,$gaps,\@array);
}
sub process_stats(){
my $sequences_ref = shift;
my $total = shift;
my $genome_size = int(0);
my $mean = $total / scalar(@$sequences_ref);
my ($n50,$n10,$n25,$n50_length,$n10_length,$n25_length,$scaffolds,$scaffolds_size,$smallest,$largest,$sequence_number,$sum);
print "Sorting...";
my @sequences=sort{$b<=>$a} @$sequences_ref;
$smallest=$sequences[-1];
$largest=$sequences[0];
print " done!\n";
$|=0;
if (($mean < 1000 && !$isnot_reads) || $is_reads ){
print "Reads detected. Ignoring N* calculations.\n";
$genome_size=$total;
$sequence_number = scalar(@$sequences_ref);
return ($mean,$n50,$n10,$n25,$n50_length,$n10_length,$n25_length,$scaffolds,$scaffolds_size,$smallest,$largest,$sequence_number,$sum) if $mean <1000;
}
elsif (!$user_genome_size){
print "Setting genome size for N* calculations to total consensus $total\n";
$genome_size=$total;
}elsif($user_genome_size){
print "Setting genome size for N* calculations to user defined $user_genome_size\n";
$genome_size = $user_genome_size;
}
foreach my $sequence_length ( @sequences){
$sum+=$sequence_length;
$sequence_number++;
if($sum >= $genome_size*0.1 && !$n10){
$n10=$sequence_number;
$n10_length=$sequence_length;
}
elsif($sum >= $genome_size*0.25 && !$n25){
$n25=$sequence_number;
$n25_length=$sequence_length;
}
elsif($sum >= $genome_size*0.5 && !$n50){
$n50 = $sequence_number;
$n50_length=$sequence_length;
}elsif ($sum >= $genome_size && !$scaffolds){
$scaffolds = $sequence_number;
$scaffolds_size = $sequence_length;
}
}
print "Processed $sequence_number sequences\n";
return ($mean,$n50,$n10,$n25,$n50_length,$n10_length,$n25_length,$scaffolds,$scaffolds_size,$smallest,$largest,$sequence_number,$sum,$genome_size);
}
sub thousands($){
my $val = shift;
return int(0) if !$val;
$val = sprintf("%.0f", $val);
1 while $val =~ s/(.*\d)(\d\d\d)/$1,$2/;
return $val;
}
+58
View File
@@ -0,0 +1,58 @@
################################################################################################################
######################## README file ########################
######################## Trinity PBS job submission with multi part dependencies ########################
######################## Author: Josh Bowden, Alexie Papanicolaou, CSIRO ########################
######################## Email: alexie@butterflybase.org ########################
######################## Version 1.0 ########################
################################################################################################################
DESCRIPTION: BASH shell scripts for submission of Trinity jobs to clusters that use PBS Torque or PBS Pro.
The set of scripts stages the parts of the Trinity workflow into 6 stages:
1/ Inchworm
2/ Chrysalis::GraphFromFasta and Chrysalis::ReadsToTranscripts if walltime permits
3/ Chrysalis::ReadsToTranscripts
4/ Chrysalis::QuantifyGraph
5/ Butterfly
6/ Gather together resulting transcripts into "Trinity.fasta" file
Each stage is submitted as a PBS job, with dependencies i.e. the following stage will only execute after successfull completion of the stage before.
Due to their parallel nature, stages 4 and 5 are submitted as 'array jobs', with each job made up of a user defined number of subtasks.
ADMINISTRATION setup instructions:
trinity_pbs script install instructions:
1. Copy all trinity_pbs.* files into a directory (we will call it "TRINITY_PBS_DIR") in a user accessible, read only, area.
2. Add TRINITY_PBS_DIR to the PATH i.e. export or set PATH=TRINITY_PBS_DIR:$PATH
3. Make sure trinity_pbs.sh is executable (chmod 755 trinity_pbs.sh)
In the file trinity_pbs.sh, do the following:
4. Change the "TRINITYPATH" variable to point to the trinity.pl installation directory.
5. Set MEMDIRIN to the name of a node-local filesystem so a network drive is not needed for final parallel stages
6. Set MODTRINITY so that the trinity executables will be available - load approprite modules and set PATH.
7. Set PBSTYPE to --pbspro or --pbs, dependent on the system present.
That should be all that is needed from an admin perspective. These scripts have been tested on PBS Torque 3.0.6 and PBS Pro 11.0.2.
There may be PBS system specific changes due to PBS version incompatabilities.
8. Maybe make any system-specific changes to the user-specific text README below or TRINITY.CONFIG.template before distributing
USER setup instructions:
1. Users should make a copy of TRINITY.CONFIG.template (possibly a copy for each job they want to run) and then modify variables in it
to suit the system the job is being run on (number of CPUs, amount of memory) and the expected runtime of the Trinity process
(which is a function of the datset size). Further instructions are provided within the TRINITY.CONFIG.template file.
2. While it is running, you can use qsub -u $USER to see your jobs
3. We provide a script, pbs_check.pl, that allows you to see the progress of your jobs; just give the job id
4. If any step fails, then some jobs that depended on a /successful/ completion of that step will be stuck in a HOLD (H) status
5. The script trinity_kill.pl can be used to kill all running, queued or held jobs by passing it the output data directory (OUTPUTDIR).
6. When you re-submit a job with trinity_pbs.sh <config file>, trinity_kill.pl will automatically run and stop and jobs in the directory specified by
your config file
(we take no responsibility if you delete the chrysalis directory and Trinity.fasta does not have all the data - e.g. because the PBS crashed)
USER recommendations
* Walltimes depend on how much data you have and also on your local HPC environment well (e.g. speed of I/O - hard disks and CPU configuration).
In the beginning you may want to err towards higher walltimes, ask colleagues using the same machines. The default values worked well for us.
* For quantify graph/Butterfly (steps 4/5), we start with a particular walltime (say 2h). Often not all jobs complete. Because it is a batch of many commands,
the best approach is to simply re-launch trinity_pbs and it will continue to process the rest of the commands. Check the logfile output from the PBS
to see if the job fails because one particular quantifygraph or Butterfly job takes longer than the given walltime (in that case: increase the walltime, run the command
manually or decrease the number of reads used - -max_reads for quantifygraph). It is possible that it is a very long gene, a bacterial contaminant or
some other oddity that is delaying your assembly.
* Do not overload the I/O of the system (steps 4/5) by starting a lot of jobs (i.e. multiple trinity assemblies). In that case, increase the
number of commands run within each step 4/5 batch (NUMPERARRAYITEM variable in the CONFIG)
* NB: Always count how many sequences you get at the end (Trinity.fasta) to make sure that the script has completed as you expected.
@@ -0,0 +1,122 @@
##################################################################################################################################
########################## ########################################
########################## Trinity PBS job submission with multi part dependencies ########################################
########################## User modifiyable input file ########################################
########################## ########################################
##################################################################################################################################
### Author: Josh Bowden, CSIRO IM&T, Alexie Papanicolaou CSIRO CES
### Email: alexie@butterflybase.org
### Version 1.0
###
### Configuration file for script to split the Trinity workflow into multiple stages so as to efficiently request
### and use appropriate resources (walltime and number of cores) on a computer cluster / supercomputer.
###
### User must set all the variables in this file to appropriate values
### and then run the trinity_pbs.sh script with this file as input as shown below:
###
### Command line usage:
### To start (or re-start) an analysis:
### >trinity_pbs.sh TRINITY.CONFIG.template
### To stop previously started PBS jobs on the queue:
### >trinity_pbs.sh --rm OUTPUTDIR
### Where:
### TRINITY.CONFIG.template = user specific job details (i.e. the current file)
### OUTPUTDIR = is path to output data directory (set below)
###
### If any stage fails, the jobs may be resubmitted and only the scripts that have not completed
### will be resubmitted to the batch system. Either the scripts can be re-run (by using trinity_pbs.sh)
### with original (or new) inputs from the current file (changing the variables below)
### or the original scripts created by trinity_pbs.sh can be re-run by finding them in the output directory.
###
### Each stage and each array job will have a PBS output file sent to the output directory when the job finishes.
### This means there will be many output files from the PBS system when the array job runs (from part 4 and 5 mostly).
### If any part fails, errors will be specified in these output files.
###
### N.B. The trinity_pbs.sh file must have a number of system specific variables set by a system administrator
###
##################################################################################################################################
# USER must edit these:
###### Set an email to which job progress and status will be sent to.
UEMAIL=
###### Set a valid account (if available), otherwise leave blank.
ACCOUNT="#PBS -A sf-CSIRO"
###### Select a value for JOBPRFIX that is not longer than 7 characters. NO spaces or other non-alphanumeric characters
JOBPREFIX=
###### Set output data directory (OUTPUTDIR)
###### OUTPUTDIR is where PBS scripts will be written and also Trinity results will be stored
###### This area requires a large amount of space (possibly 100's of GB) and a high file count
###### ($WORKDIR is a standard area on some systems, however users should check that it is valid on their machine)
OUTPUTDIR="$WORKDIR"/trinityrnaseq/"$JOBPREFIX"
###### Set input data directory. This has to be explicitly set as it is used in other internal scripts.
###### Make sure you include the final forward slash. Defaults to current directory
DATADIRECTORY=$PWD/
###### Set input filenames, you can use wildcards if you embed the filename is 'single quotes'
FILENAMELEFT='*_left.fasta' # change this
FILENAMERIGHT='*_right.fasta' #change this
FILENAMESINGLE=single.fasta # change this or set it to empty
SEQTYPE=fa # change this to fa (FASTA) fq (FASTQ) or cfa/cfq (FASTA or FASTQ colour space SOLiD ABI)
# User may opt to change these:
# we set --max_reads_per_graph to 1million because very high I/O is needed otherwise. It is unlikely that a transcript needs more than 1 million reads to be assembled....
FILENAMEINPUT=" --seqType "$SEQTYPE" --left "$DATADIRECTORY""$FILENAMELEFT" --right "$DATADIRECTORY""$FILENAMERIGHT" --max_reads_per_graph 1000000 "
### STANDARD_JOB_DETAILS sets analysis specific input to Trinity.pl
### N.B. do not use --CPU or JM flag as this is automatically appended
STANDARD_JOB_DETAILS="Trinity.pl "$FILENAMEINPUT" --output "$OUTPUTDIR""
# This is where you specify resource limits. We provide some defaults values
# Ultimately settings depends on your data size and complexity
# Steps that go beyond their walltime will not complete. Edit these values and resubmit
### Stage P1: Time and resources required for Inchworm stage
### Only use at maximum, half the available CPUs on a node
# - Inchworm will not efficiently use any more than 4 CPUs and you will have to take longer for resources to be assigned
WALLTIME_P1="2:00:00"
MEM_P1="20gb" # will use it for --JM
NCPU_P1="4"
PBSNODETYPE_P1="any" # ask you system administrator what Nodetypes exists
### Stage P2: Time and resources required for Chrysalis stage
### Starts with Bowtie alignment and post-processing of alignment file
### All CPUs presenct can be used for the Chrysalis parts.
#They may take a while to be provisioned, so the less request, possibly the faster the jobs turnaround.
# For one step (the parallel sort) it needs as much memory as specified in P1. Less memory, means more I/O for sorting
# increase for more lanes of data: 3 lanes-> 60gb of RAM and 24h of time will do it.
# The bowtie step will take considerable amount of time with more data
WALLTIME_P2="12:00:00"
MEM_P2="20gb" # will use it for the parallel sort of the SAM after alignment
NCPU_P2="6"
PBSNODETYPE_P2="any"
### Stage P3: This is a backup stage for Chrysalis - only runs if time ran out in P2 above.
### This will need about 1 day per lane
WALLTIME_P3="18:00:00"
MEM_P3="8gb"
NCPU_P3="6"
PBSNODETYPE_P3="medium"
### Stage P4: QuantifyGraph graph runs in many parallel parts
### Tasks that fail will reamain in the OUTPUTDIR/chrysalis/quantifyGraph_commands.XYZ files (XYZ is a number)
### The remaining tasks can be run by running the job submission command "trinity_pbs.sh <config_file>" again.
NUMPERARRAYITEM_P4=5000
WALLTIME_P4="00:30:00"
MEM_P4="4gb"
NCPU_P4="1"
PBSNODETYPE_P4="medium"
### Stage P5: Butterfly options. Some butterfly jobs can take exceedingly long. Users may need to restart trinity_pbs multiple times to complete.
### Tasks that fail will reamain in the OUTPUTDIR/chrysalis/butterfly_commands.adj.XYZ files (XYZ is a number)
### Often there will be a few tasks that take a lot longer than others so multiple submissions to the cluster may be required.
### The remaining tasks can be run by running the job submission command "trinity_pbs.sh <config_file>" again.
### Running the long running tasks seperately may also be a good option.
NUMPERARRAYITEM_P5=5000
WALLTIME_P5="01:00:00"
MEM_P5="10gb"
NCPU_P5="1"
PBSNODETYPE_P5="medium"
@@ -0,0 +1,100 @@
#!/usr/bin/env perl
=pod
=head1 NAME
Monitor a PBS job progress
=head1 USAGE
<job identifier> [options]
Options:
one of
-top use top on execution host
-dump dump output or error (default)
-follow follow output/error
-tail tail of output/error (the last 10 lines)
-head head of output/error (the first 10 lines)
and also
-e|error Show stderr instead of stdout
-s|spool Location of spool directory (defaults to /var/spool/PBS/spool)
=head1 LICENSE
Released under the MIT License - Alexie Papanicolaou 2012, CSIRO Ecosystem Sciences, alexie@butterflybase.org
=cut
use strict;
use warnings;
use Pod::Usage;
use Getopt::Long;
my ($head,$follow,$tail,$dump,$show_error,$spool_dir,$top);
GetOptions(
'top' => \$top,
'head' => \$head,
'follow' => \$follow,
'tail' => \$tail,
'cat|dump' => \$dump,
's|spool_dir:s' => \$spool_dir,
'error' => \$show_error,
);
my $method;
if ($top){
$method = 'top';
}elsif ($head){
$method = 'head';
}elsif ($follow){
$method = 'follow'
}elsif ($tail){
$method = 'tail';
}else{
$method = 'cat';
}
my $jobid = shift;
pod2usage unless $jobid;
$spool_dir = $spool_dir ? $spool_dir : '/var/spool/PBS/spool'; # exists on exec host but necessarily on submit host
my $user = $ENV{'USER'};
my $exec = 'ssh ';
pod2usage "No job ID provided\n" unless $jobid;
$jobid=~/^(\w+\[?\d*\]?)/;
$jobid=$1 || pod2usage "Not a valid job ID $jobid\n";
my $pbs_server = `qstat -Bf|grep ^Server`;
chomp($pbs_server);
$pbs_server=~s/^Server: //;
die "No PBS server found. Is it online?\n" unless $pbs_server;
my $node=`qstat -f $jobid|grep exec_host`;
$node =~/exec_host\s+=\s+(\w+)/;
$node = $1 || "No valid host found. Is $jobid a live job?\n";
my $cmd;
if ($method=~/^c/){
$cmd = 'cat';
}elsif ($method=~/^ta/){
$cmd = 'tail';
}elsif ($method=~/^f/){
$cmd = 'tail -f';
}elsif ($method=~/^h/){
$cmd = 'head';
}else {
$cmd = $method;
}
if ($method eq 'top'){
system("ssh $node -t top");
}else{
$exec.= " $node $cmd $spool_dir/$jobid.$pbs_server.OU" if !$show_error;
$exec.= " $node $cmd $spool_dir/$jobid.$pbs_server.ER" if $show_error;
system($exec);
}
print "\n#\tQPEEK complete for job $jobid. Host was: $node\n";
@@ -0,0 +1,36 @@
#!/usr/bin/env perl
use strict;
use warnings;
my $me = $ENV{'USER'};
my @jobs_running_ln = `qstat -u $ENV{'USER'}`;
if (!@jobs_running_ln || scalar(@jobs_running_ln)<1){
print "No running jobs for user $me\n";
exit();
}
my %jobs_running;
foreach my $job_ln (@jobs_running_ln){
$job_ln=~/^(\S+)/;
$jobs_running{$1}=1 if $1;
}
my $dir=$ARGV[0] ? $ARGV[0] : '.';
if ($dir && -d $dir){
# read jobnumbers in current directory (and directory passed as variable) and kill them (start at the last job and move to oldest)
if (-s $dir."/jobnumbers.out"){
my @job_sub = `tac $dir/jobnumbers.out`;
chomp(@job_sub);
foreach my $job (@job_sub){
system("qdel -W force $job") if $jobs_running{$job};
# twice to make sure
system("qdel -W force $job >/dev/null 2>/dev/null") if $jobs_running{$job};
}
unlink($dir."/jobnumbers.out");
}
else{
print "No previous jobs found\n";
}
}
@@ -0,0 +1,31 @@
#!/bin/bash
# much slower than perl version
# read jobnumbers in current directory (and directory passed as variable) and kill them (start at the last job and move to oldest)
if [ $1 ]; then
if [ -e "$1/jobnumbers.out" ];then
tac "$1/jobnumbers.out" |
while read line
do
echo "$1/jobnumbers.out": Stopping $line
qdel -W force $line
qdel -W force $line >/dev/null 2>/dev/null
done
rm -f "$1/jobnumbers.out"
fi
exit 0
fi
if [ -e jobnumbers.out ];then
tac jobnumbers.out |
while read line
do
echo jobnumbers.out: Stopping $line
qdel -W force $line
qdel -W force $line >/dev/null 2>/dev/null
done
rm -f jobnumbers.out
else
echo "No previous jobs found"
fi
@@ -0,0 +1,70 @@
##################################################################################################################################
########################## ########################################
########################## Trinity PBS job submission with multi part dependencies ########################################
########################## ########################################
##################################################################################################################################
##################################################################################################################################
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
### Version 1.0
########### The main control script, that we will run and can be run seperately later if the job fails at intermediate stages ###################
############################################################################################################################################################
RUNVAR1=""$HASHBANG"
echo Killing any running jobs in "$OUTPUTDIR"
trinity_kill.pl "$OUTPUTDIR" 2> /dev/null >/dev/null
echo Checking and submitting any jobs
###########################################################################################################################################################
################ Use the Trinity/Chrysalis checkpoint files to work out what part of job remains #########################################
################ and queue only those parts that are still required #########################################
if [ -s \""$OUTPUTDIR"/Trinity.fasta\" ] ; then
echo 'Trinity seemingly finished: Trinity.fasta present in output directory. If you think this is wrong, delete Trinity.fasta and re-submit'
exit 0
fi
if [ ! -e \""$OUTPUTDIR"/inchworm.K25.L25"$DS"fa.finished\" ] ;then # do the whole analysis
PBS_JOB1=\`qsub "$JOBNAME1".sh\`
PBS_JOB2=\`qsub -W depend=afterok:\$PBS_JOB1 "$JOBNAME2".sh\`
PBS_JOB3=\`qsub -W depend=afternotok:\$PBS_JOB2 "$JOBNAME3".sh\`
PBS_JOB4_2=\`qsub -W depend=afterok:\$PBS_JOB2,afterany:\$PBS_JOB3 "$JOBNAME4".sh\`
echo "$JOBNAME1".sh submitted ; echo \"\$PBS_JOB1\" > jobnumbers.out ;
echo "$JOBNAME2".sh submitted ; echo \"\$PBS_JOB2\" >> jobnumbers.out ;
echo "$JOBNAME3".sh submitted ; echo \"\$PBS_JOB3\" >> jobnumbers.out ;
echo "$JOBNAME4".sh post-job "$JOBNAME2" submitted ; echo \"\$PBS_JOB4_2\" >> jobnumbers.out ;
else # do analysis after inchworm only
if [ ! -e \""$OUTPUTDIR"/chrysalis/GraphFromIwormFasta.finished\" ] ; then # start analysis after inchworm
PBS_JOB2=\`qsub "$JOBNAME2".sh\`
PBS_JOB3=\`qsub -W depend=afternotok:\$PBS_JOB2 "$JOBNAME3".sh\`
PBS_JOB4_2=\`qsub -W depend=afterok:\$PBS_JOB2,afterany:\$PBS_JOB3 "$JOBNAME4".sh\`
echo "$JOBNAME2".sh submitted ; echo \"\$PBS_JOB2\" > jobnumbers.out ;
echo "$JOBNAME3".sh submitted ; echo \"\$PBS_JOB3\" >> jobnumbers.out ;
echo "$JOBNAME4".sh post-job "$JOBNAME2" submitted ; echo \"\$PBS_JOB4_2\" >> jobnumbers.out ;
else
if [ ! -e \""$OUTPUTDIR"/chrysalis/readsToComponents.finished\" ] ;then # start analysis at Chrisyalis ReadsToTranscripts - which is slow due to I/O
PBS_JOB3=\`qsub "$JOBNAME3".sh\`
PBS_JOB4=\`qsub -W depend=afterok:\$PBS_JOB3 "$JOBNAME4".sh\`
echo "$JOBNAME3".sh submitted ; echo \"\$PBS_JOB3\" > jobnumbers.out ;
echo "$JOBNAME4".sh post-job "$JOBNAME3" submitted ; echo \"\$PBS_JOB4\" >> jobnumbers.out ;
else # Run Chrysalis QuantifyGraph and then Butterfly
PBS_JOB4=\`qsub "$JOBNAME4".sh\`
echo "$JOBNAME4".sh submitted ; echo \"\$PBS_JOB4\" > jobnumbers.out ;
fi
fi
echo When JOB ID \"\$PBS_JOB4\" finishes successfully - see output: \"$OUTPUTDIR\"/\"$JOBNAME4\".o\"\$PBS_JOB4\" - either re-run the submit command or use the following command to get all the data into a single file:
echo ' find "$OUTPUTDIR"/chrysalis -name *allProbPaths.fasta -exec cat {} \\; > "$OUTPUTDIR"/Trinity.fasta '
fi
"
echo "${RUNVAR1}" | cat -> ""$JOBPREFIX"_run.sh"
chmod 744 ""$JOBPREFIX"_run.sh"
echo "To restart these jobs run either the same command again:"
echo " trinity_pbs.sh <the same .config file> "
echo " or the following script: "
echo " "$JOBPREFIX"_run.sh found in the output directory "$OUTPUTDIR" "
echo "To stop these jobs run:"
echo " trinity_kill.pl "$OUTPUTDIR" "
echo "To check progress of these jobs run:"
echo " qstat -u "$USER""
echo ""
@@ -0,0 +1,179 @@
##################################################################################################################################
########################## ########################################
########################## Trinity PBS job submission with multi part dependencies ########################################
########################## ########################################
##################################################################################################################################
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
### Version 1.0
### Function definitions and PBS version specific information
##################################################################################################################################
######################################################################################################################
### Return usage information if user requests --help -h or another incorrect input flag.
######################################################################################################################
function F_USAGE {
echo ""
echo " Usage: "
echo ""
echo " To start an analysis: "
echo " trinity_pbs.sh <config.file> "
echo ""
echo " To stop previously started PBS jobs in the queue: "
echo " trinity_kill.pl OUTPUTDIR "
echo ""
echo " Where:"
echo " <config.file> = contains specific job details (see below)"
echo " OUTPUTDIR = is path to output data directory"
echo ""
echo " Note: Modify the values in CONFIG_FILE to suit compute cluster and "
echo " input data size and nameing conventions:"
echo ""
echo " <config.file> contains data and job input details and user defined variables "
echo " and user and cluster specific options:"
echo " UEMAIL, DATADIRECTORY, OUTPUTDIR, JOBPREFIX, ACCOUNT "
echo " STANDARD_JOB_DETAILS"
echo " per job details:"
echo " NCPU, WALLTIME, MEM, MEMDIR, NUMPERARRAYITEM"
echo ""
}
######################################################################################################################
### Function to return the PBS header information, depending on PBS type.
######################################################################################################################
# NODESCPUS=$(F_GETNODESTRING $1 $MEM $NCPU large $WALLTIME $JOBNAME $ACCOUNT $PBSUSER $MODTRINITY $JOBPREFIX)
# NODESCPUS=$(F_GETNODESTRING $1 $2 $3 $4 $5 $6 $7 $8 $9 ${10} )
function F_GETNODESTRING {
if [[ $1 = "--pbspro" ]] ; then
echo "
#PBS -l select=1:ncpus="$3":NodeType="$4":mem="$2"
#PBS -l walltime="$5"
#PBS -N "$6"
"$7"
#PBS -j oe
#PBS -m a
#PBS -V
"$8"
"$9"
"
elif [[ $1 = "--pbs" ]] ; then
echo "
#PBS -l nodes=1:ppn="$3"
#PBS -l vmem="$2"
#PBS -l walltime="$5"
#PBS -N "$6"
#PBS -j oe
#PBS -m a
"$8"
#PBS -V
"$9"
"
fi
echo " JOBNAME=$6"
echo " JOBPREFIX=${10}"
}
######################################################################################################################
### Do some checking that files exist etc.
######################################################################################################################
#DS=$(F_GETNODESTRING $FILENAMEINPUT $DATADIRECTORY $FILENAMESINGLE $FILENAMELEFT $FILENAMERIGHT)
# DS=$(F_GETNODESTRING $1 $2 $3 $4 $5 )
function SET_DS {
# add the correct filename extension so we can check if inchworm has been run in part 3 below
if [[ "$1" == *FR* ]] || [[ "$1" == *RF* ]] ;
then
DS="."
else
DS=".DS."
fi
}
#AP: i removed this because a) it didn't stop the script from progressing when there was an error, b) it was not handling multiple input files
function F_CHECKFILES {
# Check that input data file(s) exist.
if [[ "$1" == *--single* ]] ; then
if [ ! -f ""$2""$3"" ] ; then
echo "Input file does not exist: "
echo " "$2""$3""
exit 1
fi
else # check that both left and right filenames exist
if [ ! -f ""$2""$4"" ] ; then
echo "Input file does not exist: "
echo " "$2""$4""
exit 1
fi
if [ ! -f ""$2""$5"" ] ; then
echo "Input file does not exist: "
echo " "$2""$5""
exit 1
fi
fi
}
######################################################################################################################
### Checking that files containing stage specific script data exist
######################################################################################################################
#F_WRITESCRIPT( scriptname $SOURCENAME )
#F_WRITESCRIPT( $1 $2 )
function F_WRITESCRIPT {
if [ -e "$2" ] ; then
source "$2"
else
echo ""$1" requires file \""$2"\" to be present in the current directory"
exit 1
fi
}
##################################################################################################################################
# This is used to stop all jobs sent to the PBS queue
# requires path to the filename as second argument
# The order of the following tests is important.
##################################################################################################################################
if [[ "$1" = "--rm" ]] || [[ "$1" = "-rm" ]] ; then
if [[ -e $2/jobnumbers.out ]] ; then
cat $2/jobnumbers.out | while read LINE; do
qdel $LINE || { echo "continuing..." ;}
done
#echo " Any parallel "
else
echo " Could not stop jobs as could not open file: "
echo " $2"jobnumbers.out""
fi
exit 0
fi
if [[ "$1" = "--help" ]] || [[ "$1" = "-h" ]] || [[ "$1" = "-?" ]]; then
F_USAGE
exit 0
fi
##################################################################################################################################
## User email information required from command line:
## Provide a valid email address so PBS can email user on start and finish of execution of individual parts (not part 4b or 5b though)
##################################################################################################################################
if [[ ! -e "$1" ]] ; then
echo "input file does not exist: "$1""
echo "Please provide an input file"
F_USAGE
exit 0
fi
@@ -0,0 +1,29 @@
##################################################################################################################################
########################## ########################################
########################## Trinity PBS job submission with multi part dependencies ########################################
########################## ########################################
##################################################################################################################################
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
### Version 1.0
### Inchworm P1 script
##################################################################################################################################
if [[ $MEM_P1 =~ ^([0-9]+) ]]; then
let MEM_BASE="${BASH_REMATCH[1]}"
JM_MEM="$MEM_BASE"G
else
echo No memory given for kmer counter: "$MEM_P1"
exit 1
fi
JOBSTRING1=""$HASHBANG"
"$NODESCPUS"
cd "$OUTPUTDIR"
export OMP_NUM_THREADS="$NCPU_P1"
export KMP_AFFINITY=compact
# this runs Inchworm only
"$STANDARD_JOB_DETAILS" --JM "$JM_MEM" --CPU "$NCPU_P1" --no_run_chrysalis
"
# Write the JOBSTRING1 to a file for later execution
echo "${JOBSTRING1}" | cat -> ""$JOBNAME1".sh"
@@ -0,0 +1,43 @@
##################################################################################################################################
########################## ########################################
########################## Trinity PBS job submission with multi part dependencies ########################################
########################## ########################################
##################################################################################################################################
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
### Version 1.0
### Chrysalis P2 script
##################################################################################################################################
if [[ $MEM_P2 =~ ^([0-9]+) ]]; then
let MEM_BASE="${BASH_REMATCH[1]}"
let ALIGN_MEM=$MEM_BASE-5
if [ $ALIGN_MEM -le 0 ];then
let ALIGN_MEM=$MEM_BASE-2
if [ $ALIGN_MEM -le 0 ];then
echo "Memory requested for MEM_P2 is too low. Ask for at least 5 gigabytes"
exit 1
fi
fi
ALIGN_MEM="$ALIGN_MEM"G
else
echo No memory given: "$MEM_P2"
exit 1
fi
JOBSTRING2=""$HASHBANG"
"$NODESCPUS"
cd "$OUTPUTDIR"
# set stack size to unlimited for Chrysalis (part 1)
ulimit -s unlimited
export OMP_NUM_THREADS="$NCPU_P2"
export KMP_AFFINITY=scatter
# this runs Chrysalis::GraphFromFasta and maybe Chrysalis::GraphFromFasta if there is still walltime
"$STANDARD_JOB_DETAILS" --JM "$ALIGN_MEM" --CPU "$NCPU_P2" --no_run_quantifygraph
"
# Write the JOBSTRING2 to a file for later execution
echo "${JOBSTRING2}" | cat -> ""$JOBNAME2".sh"
@@ -0,0 +1,38 @@
##################################################################################################################################
########################## ########################################
########################## Trinity PBS job submission with multi part dependencies ########################################
########################## ########################################
##################################################################################################################################
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
### Version 1.0
### Chrysalis P3 script (only run if Chrysalis P3 has not completed)
##################################################################################################################################
if [[ $MEM_P3 =~ ^([0-9]+) ]]; then
let MEM_BASE="${BASH_REMATCH[1]}"
let ALIGN_MEM=$MEM_BASE-5
if [ $ALIGN_MEM -le 0 ];then
let ALIGN_MEM=$MEM_BASE-2
if [ $ALIGN_MEM -le 0 ];then
echo "Memory requested for MEM_P3 is too low. Ask for at least 5 gigabytes"
exit 1
fi
fi
ALIGN_MEM="$ALIGN_MEM"G
else
echo No memory given: "$MEM_P3"
exit 1
fi
JOBSTRING3=""$HASHBANG"
"$NODESCPUS"
cd "$OUTPUTDIR"
ulimit -s unlimited
export OMP_NUM_THREADS="$NCPU_P3"
export KMP_AFFINITY=scatter
# this runs Chrysalis::ReadsToTranscripts if it has not completed in the previous step
"$STANDARD_JOB_DETAILS" --JM "$ALIGN_MEM" --CPU "$NCPU_P3" --no_run_quantifygraph
"
# Write the JOBSTRING3 to a file for later execution
echo "${JOBSTRING3}" | cat -> ""$JOBNAME3".sh"
@@ -0,0 +1,156 @@
##################################################################################################################################
########################## ########################################
########################## Trinity PBS job submission with multi part dependencies ########################################
########################## ########################################
##################################################################################################################################
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
### Version 1.0
### QuantifyGraph and Butterfly p4a Script
##################################################################################################################################
# we will not use array in order to ensure only jobs that have not finished are Submitting.
# this does cause a problem when wanting to kill them manually but best to use kill script
## JOBPREFIX is passed via HASHBANG
JOBSTRING4=""$HASHBANG"
"$NODESCPUS"
JOB_CHRYSALIS="$JOBPREFIX"_p4b
JOB_BUTTERFLY="$JOBPREFIX"_p5b
cd "$OUTPUTDIR"
rm -f \$JOB_CHRYSALIS.jobnames \$JOB_BUTTERFLY.jobnames
FILENAME=""$OUTPUTDIR"/chrysalis/quantifyGraph_commands"
FILENAMEBFLY=""$OUTPUTDIR"/chrysalis/butterfly_commands"
if [ ! -e \$FILENAME.pbs ];then
sed -e 's/.*/if [[ -e SEDPLACEHOLDER ]]; then &;fi/' \$FILENAME|sed -r 's/(-e\s)SEDPLACEHOLDER(\s.+-i\s)(\S+).tmp/\1\3.tmp\2\3.tmp/' > \$FILENAME.pbs
split -d -a 3 -l "1000" \$FILENAME.pbs \$FILENAME.pbs.
sleep 5
fi
if [ ! -e \$FILENAMEBFLY.pbs ];then
sed -e 's/.*/if [ -e SEDPLACEHOLDER ]; then &;fi/' \$FILENAMEBFLY|sed -r 's/(-e\s)SEDPLACEHOLDER(\s.+-C\s)(\S+)/\1\3.out\2\3/' > \$FILENAMEBFLY.pbs
split -d -a 3 -l "1000" \$FILENAMEBFLY.pbs \$FILENAMEBFLY.pbs.
sleep 5
fi
FILENAME=\$FILENAME.pbs
FILENAMEBFLY=\$FILENAMEBFLY.pbs
NUMCMDS=\`ls -l \$FILENAME.??? | wc -l\`
NUMCMDSBFLY=\`ls -l \$FILENAMEBFLY.??? | wc -l\`
let NUMCMDS=\$NUMCMDS-1
let NUMCMDSBFLY=\$NUMCMDSBFLY-1
let SUBMITTED_C=0
let SUBMITTED_B=0
for ((JOBID=0;JOBID<=\$NUMCMDS;++JOBID));do
JOB_INDEX_PADDED=\`printf "%03d" \$JOBID\`
MYJOBQ=\""\$FILENAME".\$JOB_INDEX_PADDED\"
MYJOBB=\""\$FILENAMEBFLY".\$JOB_INDEX_PADDED\"
JOB_FILESIZE_Q=\$(stat -c%s \$MYJOBQ)
JOB_FILESIZE_B=\$(stat -c%s \$MYJOBB)
#if some Q have completed:
if [ -s \"\$MYJOBQ.completed\" ] ; then
JOB_COMPLETED_FILESIZE_Q=\$(stat -c%s \"\$MYJOBQ.completed\")
# if not all have completed then run both Q and B
if [ \"\$JOB_FILESIZE_Q\" -gt \"\$JOB_COMPLETED_FILESIZE_Q\" ] ; then
PBS_JOB4=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_CHRYSALIS.sh\" \`
if [[ ! \$PBS_JOB4 ]]; then
echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_CHRYSALIS.sh\" FAILED. Aborting...\"
exit 255
fi
let SUBMITTED_C++
echo \$PBS_JOB4 >> \"\$JOB_CHRYSALIS.jobnames\"
echo \$PBS_JOB4 >> jobnumbers.out ;
PBS_JOB5=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED -W depend=afterok:$PBS_JOB4 \"\$JOB_BUTTERFLY.sh\" \`
if [[ ! \$PBS_JOB5 ]]; then
echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED -W depend=afterok:$PBS_JOB4 \"\$JOB_BUTTERFLY.sh\" FAILED. Aborting...\"
exit 255
fi
let SUBMITTED_B++
echo \$PBS_JOB5 >> \"\$JOB_BUTTERFLY.jobnames\"
echo \$PBS_JOB5 >> jobnumbers.out ;
if [ \$(( \$JOBID % 20 )) -eq 0 ] ; then
echo Submitting up to 20 Quantify and/or Butterfly jobs
sleep 3 # be nice
fi
# else all Q have completed; have B completed?
else
# if at least some B have completed
if [ -s \"\$MYJOBB.completed\" ] ; then
JOB_COMPLETED_FILESIZE_B=\$(stat -c%s \"\$MYJOBB.completed\" )
# if not all, run them with no dependency (Q has completed)
if [ \"\$JOB_FILESIZE_B\" -gt \"\$JOB_COMPLETED_FILESIZE_B\" ] ; then
PBS_JOB5=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_BUTTERFLY.sh\" \`
if [[ ! \$PBS_JOB5 ]]; then
echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_BUTTERFLY.sh\" FAILED. Aborting...\"
exit 255
fi
let SUBMITTED_B++
echo \$PBS_JOB5 >> \"\$JOB_BUTTERFLY.jobnames\"
echo \$PBS_JOB5 >> jobnumbers.out ;
if [ \$(( \$JOBID % 20 )) -eq 0 ] ; then
echo Submitting up to 20 Quantify and/or Butterfly jobs
sleep 3 # be nice
fi
fi
# else no Q have completed; run them without dependency
else
PBS_JOB5=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_BUTTERFLY.sh\" \`
if [[ ! \$PBS_JOB5 ]]; then
echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_BUTTERFLY.sh\" FAILED. Aborting...\"
exit 255
fi
let SUBMITTED_B++
echo \$PBS_JOB5 >> \"\$JOB_BUTTERFLY.jobnames\"
echo \$PBS_JOB5 >> jobnumbers.out ;
if [ \$(( \$JOBID % 20 )) -eq 0 ] ; then
echo Submitting up to 20 Quantify and/or Butterfly jobs
sleep 3 # be nice
fi
fi
fi
# neither Q (and thus nor B) have ever ran successfully, submit both with a dependency
else
PBS_JOB4=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_CHRYSALIS.sh\" \`
if [[ ! \$PBS_JOB4 ]]; then
echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_CHRYSALIS.sh\" FAILED. Aborting...\"
exit 255
fi
let SUBMITTED_C++
echo \$PBS_JOB4 >> \"\$JOB_CHRYSALIS.jobnames\"
echo \$PBS_JOB4 >> jobnumbers.out ;
PBS_JOB5=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED -W depend=afterok:\$PBS_JOB4 \"\$JOB_BUTTERFLY.sh\" \`
if [[ ! \$PBS_JOB5 ]]; then
echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED -W depend=afterok:\$PBS_JOB4 \"\$JOB_BUTTERFLY.sh\" FAILED. Aborting...\"
exit 255
fi
let SUBMITTED_B++
echo \$PBS_JOB5 >> \"\$JOB_BUTTERFLY.jobnames\"
echo \$PBS_JOB5 >> jobnumbers.out ;
if [ \$(( \$JOBID % 20 )) -eq 0 ] ; then
echo Submitting up to 20 Quantify and/or Butterfly jobs
sleep 3 # be nice
fi
fi
done
echo Submitted \$SUBMITTED_C Chrysalis and \$SUBMITTED_B Butterfly jobs
if [[ \$SUBMITTED_B == 0 && \$SUBMITTED_C == 0 ]]; then
echo \"No Trinity jobs need to be submitted \"
if [ -s "$OUTPUTDIR"/Trinity.fasta.complete ]; then
echo \"Trinity RNA-Seq assembly is complete! Result file is present as "$OUTPUTDIR"/Trinity.fasta \"
else
echo \"Proceeding with capturing the output with this command\"
echo ' find "$OUTPUTDIR"/chrysalis -name *allProbPaths.fasta -exec cat {} \\; > "$OUTPUTDIR"/Trinity.fasta '
find "$OUTPUTDIR"/chrysalis -name *allProbPaths.fasta -exec cat {} \\; > "$OUTPUTDIR"/Trinity.fasta
touch "$OUTPUTDIR"/Trinity.fasta.complete
echo DO: rm -f "$OUTPUTDIR"/bowtie.nameSorted.sam* "$OUTPUTDIR"/both.fa* "$OUTPUTDIR"/inchworm.kmer_count "$OUTPUTDIR"/iworm_* "$OUTPUTDIR"/target* "$OUTPUTDIR"/jellyfish* "$OUTPUTDIR"/scaffolding* "$OUTPUTDIR"/*.finished "$OUTPUTDIR"/mer_counts_*
fi
fi
"
######
##### Write the above script to a file for later execution
echo "${JOBSTRING4}" | cat -> "$JOBNAME4.sh"
@@ -0,0 +1,42 @@
##################################################################################################################################
########################## ########################################
########################## Trinity PBS job submission with multi part dependencies ########################################
########################## ########################################
##################################################################################################################################
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
### Version 1.0
### QuantifyGraph p4b Script
##################################################################################################################################
JOBSTRING4b=""$HASHBANG"
"$NODESCPUS"
if [[ ! \$JOB_INDEX_PADDED ]];then
echo \"Error: not a proper submission\"
exit 255
fi
echo \"Processing quantifyGraph_commands index \$JOB_INDEX_PADDED \"
cd "$OUTPUTDIR"
export OMP_NUM_THREADS=1
COREFILENAME=""$OUTPUTDIR"/chrysalis/quantifyGraph_commands.pbs"
MYJOBQ=\$COREFILENAME.\$JOB_INDEX_PADDED
JOB_FILESIZE=\$(stat -c%s \"$MYJOBQ\")
if [ -s \"\$MYJOBQ.completed\" ] ; then
JOB_COMPLETED_FILESIZE=\$(stat -c%s \"\$MYJOBQ.completed\")
if [ \"\$JOB_FILESIZE\" != \"\$JOB_COMPLETED_FILESIZE\" ] ; then
trap \" echo \\\"Please check \$MYJOBQ Chrysalis QuantifyGraph processes had enough walltime.\\\"; exit 255 \" INT TERM
"$TRINITYPATH"/trinity-plugins/parafly/bin/ParaFly -CPU "$NCPU_P4" -v -failed_cmds \$MYJOBQ.failed -c \$MYJOBQ
trap - INT TERM EXIT
fi
else
trap \" echo \\\"Please check \$MYJOBQ Chrysalis QuantifyGraph processes had enough walltime.\\\"; exit 255 \" INT TERM
"$TRINITYPATH"/trinity-plugins/parafly/bin/ParaFly -CPU "$NCPU_P4" -v -failed_cmds \$MYJOBQ.failed -c \$MYJOBQ
trap - INT TERM EXIT
fi
sleep 30 # IO friendship for following butterfly job - sometimes butterfly fails to find output if io is overwhelmed
exit
"
# Write the above script to a file for later execution
echo "${JOBSTRING4b}" | cat -> "$JOBPREFIX"_p4b.sh
@@ -0,0 +1,41 @@
##################################################################################################################################
########################## ########################################
########################## Trinity PBS job submission with multi part dependencies ########################################
########################## ########################################
##################################################################################################################################
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
### Version 1.0
### Butterfly p5b Script
##################################################################################################################################
JOBSTRING5b=""$HASHBANG"
"$NODESCPUS"
if [[ ! \$JOB_INDEX_PADDED ]];then
echo \"Error: not a proper submission\"
exit 255
fi
echo \"Processing butterfly_commands index \$JOB_INDEX_PADDED \"
cd "$OUTPUTDIR"
export OMP_NUM_THREADS=1
COREFILENAME=""$OUTPUTDIR"/chrysalis/butterfly_commands.pbs"
MYJOBB=\$COREFILENAME.\$JOB_INDEX_PADDED
JOB_FILESIZE=\$(stat -c%s \"\$MYJOBB\")
if [ -s \"\$MYJOBB.completed\" ] ; then
JOB_COMPLETED_FILESIZE=\$(stat -c%s \"\$MYJOBB.completed\")
if [ \"$JOB_FILESIZE\" != \"\$JOB_COMPLETED_FILESIZE\" ] ; then
trap \" echo \\\"Please check \$MYJOBB Butterfly processes had enough walltime.\\\"; exit 255 \" INT TERM
"$TRINITYPATH"/trinity-plugins/parafly/bin/ParaFly -CPU "$NCPU_P5" -v -failed_cmds \$MYJOBB.failed -c \$MYJOBB
trap - INT TERM EXIT
fi
else
trap \" echo \\\"Please check \$MYJOBB Butterfly processes had enough walltime.\\\"; exit 255 \" INT TERM
"$TRINITYPATH"/trinity-plugins/parafly/bin/ParaFly -CPU "$NCPU_P5" -v -failed_cmds \$MYJOBB.failed -c \$MYJOBB
trap - INT TERM EXIT
fi
exit
"
# Write the above script to a file for later execution
echo "${JOBSTRING5b}" | cat -> "$JOBPREFIX"_p5b.sh
@@ -0,0 +1,291 @@
#!/bin/bash
set -e # turn on exit on error
##################################################################################################################################
########################## ########################################
########################## Trinity PBS job submission with multi part dependencies ########################################
########################## ########################################
##################################################################################################################################
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
### Version 1.0
###
###
### Script to split the Trinity workflow into multiple stages so as to efficiently request
### and use appropriate resources (walltime and number of cores) on a computer cluster / supercomputer.
### Currently creates scripts for PBS Torque or PBSpro
###
### trinity_pbs script install instructions:
### 1. Copy all trinity_pbs.* files into a directory (we will call it "TRINITY_PBS_DIR").
### 2. Add TRINITY_PBS_DIR to the PATH i.e. export or set PATH=TRINITY_PBS_DIR:$PATH (perhaps export PATH in .bashrc file)
### 3. Change the "TRINITYPBSPATH" variable found below to point to the directory also. i.e. TRINITYPBSPATH=TRINITY_PBS_DIR
### 4. Set MEMDIRIN to name of a node-local filesystem so a network drive is not needed unecesarily for Scripts 4b and 5b
### 5. Set MODTRINITY to any modules that need to be loaded so Trinity.pl can be run.
### 6. Set TRINITYPATH to the path to Trinity.pl executable
### 7. Set PBSTYPE to --pbspro or --pbs, dependent on the system present.
### That should be all that is needed from an admin perspective (besides making scripts accessible and exectable for users)
###
### Users need make a copy of TRINITY.CONFIG.template and then modify variables in it. See TRINITY.CONFIG.template for further details.
###
### The current script does the following.
### Part 1. Reads data from TRINITY.CONFIG and creates the input directory, data file names, output data directory and Trinity.pl command line
### User inputs from TRINITY.CONFIG file :
### JOBPREFIX A string of less than 11 characters long. PBS will use this as a jobname prefix.
### DATADIRECTORY Where input data exists
### OUTPUTDIR Where user wants output data to go - requires a lot of space even for small datatsets
### STANDARD_JOB_DETAILS the Trinity.pl command line
### ACCOUNT Account details of user (if required by PBS system being used)
###
### Part 2. Writes scripts to run Trinity.pl in 6 stages:
### 3 intial (Inchworm, and 2 x Chrysalis stages: Chrysalis::GraphFromFasta and Chrysalis::ReadsFromTranscripts)
### 2 parallel stages (Chrysalis::QuantifyGraph and Butterfly) which are executed in parallel.
### 1 collection of results as Trinity.Fasta.
###
### Information input from command line filename for stage 'x' :
### WALLTIME_Px Amount of time stage requires
### MEM_Px The amount of memory the stage requires
### NCPU_Px The number of CPUs the stage may use
### PBSNODETYPE _Px The PBS (for --pbspro only) queue name
### NUMPERARRAYITEM_Px The number of massively parallel jobs in each parallel satge.
###
### Part 3. Runs scripts dependant upon what stage has been detected as completed, using PBS job dependencies
###
### Command line usage:
### To start (or re-start) an analysis:
### >trinity_pbs.sh TRINITY.CONFIG.template
### To stop previously started PBS jobs on the queue:
### >trinity_kill.pl OUTPUTDIR
### Where:
### TRINITY.CONFIG.template = user specific job details
### OUTPUTDIR = is path to output data directory
###
### Output job script submission files. These are saved in the output directory (OUTPUTDIR) and can be modified/re-run if any job fails.
### *_run.sh Runs all the following scripts - with job dependencies and only the jobs that still need to be run.
### *_p1.sh Runs Inchworm stage. Does not scale well past a single socket. Only request at most the number of cores on a single CPU.
### *_p2.sh Runs Chrysalis::GrapghFromFasta clustering of Inchworm output. Should scale to number of cores on node
### *_p3.sh Runs Chrysalis::ReadsToTranscripts. I/O limited. Try to use local filesystem (not implemented)
### *_p4a.sh Creates jobs to run Chrysalis::QuantifyGraph and Butterfly parallel tasks
### *_p4b.sh QuantifyGraph job. "NUMPERARRAYITEM" tasks from the file /chrysalis/quantifyGraph_commands are run for each job
### *_p5b.sh Butterfly job. "NUMPERARRAYITEM" tasks from the file /chrysalis/butterfly_commands are run for each job
### to start off at last completed stage. At present leaves all data on temporary area of shared network drive and
### copies Trinity.fatsa to home directory (with specific job prefix in filename).
### * = $JOBPREFIX. $JOBPREFIX should not be > 10 characters long
##################################################################################################################################
####################### SET TRINITY INSTALLATION PATH (where Trinity.pl resides #################################################
##################################################################################################################################
# We need the path even if loaded using module
TRINITYPATH="/home/pap056/software/trinity_2013_08_14/"
# If you are loading using module, you can make the next variable blank
NEWPATH=
NEWPATH="export PATH=$PATH:$TRINITYPATH"
##################################################################################################################################
######## Set TRINITYPBSPATH to the directory where trinity_pbs.sh scripts are installed
##################################################################################################################################
TRINITYPBSPATH=`dirname "$0"`; # set to location of this script
##################################################################################################################################
######## Set cluster specific name for compute node local filesystem
##################################################################################################################################
MEMDIRIN="\$TMPDIR" # available on Barrine
##################################################################################################################################
######## Set system specific PATHS and load system specific modules (if available)
##################################################################################################################################
# Example for for Barrine:
PBSTYPE="--pbspro"
# Here we ensure that Java 1.6 is used and Java 1.7 is removed (Butterfly dependency)
MODTRINITY="
module load mpt/2.00 perl/5.15.8 bowtie/12.7 jellyfish/1.1.5 samtools/1.18 java/1.6.0_22-sun;
module rm java/1.7.0_02
"
## That should be all the admin modifications needed.
# Append Trinity path data to env variables loaded by every script
MODTRINITY="
$NEWPATH;
export TRINITYPATH="/home/pap056/software/trinity_2013_08_14";
$MODTRINITY
"
##################################################################################################################################
##################################################################################################################################
##################################################################################################################################
##################################################################################################################################
######### Load files that contains functions
##################################################################################################################################
# Modify function F_GETNODESTRING in file trinity_pbs.header so that a correct PBS header is returned to suit your PBS cluster
if [ -e "$TRINITYPBSPATH"/trinity_pbs.header ] ; then
source "$TRINITYPBSPATH"/trinity_pbs.header
else
echo "$1 requires file \"trinity_pbs.header\" to be present in: "
echo "$TRINITYPBSPATH"
exit 1
fi
##################################################################################################################################
######### Load input config file
##################################################################################################################################
if [ -e "$1" ] ; then
source "$1"
else
echo "Error: Input file does not exist: "$1" "
exit 1
fi
##################################################################################################################################
## Common variables to PBS and PBSpro
## and other needed variables that a user should not need to modify
##################################################################################################################################
if [ $UEMAIL ]; then PBSUSER="#PBS -M "$UEMAIL"" ; fi
HASHBANG="#!/bin/bash"
##################################################################################################################################
## PBS torque and PBSpro have some differences.
## Organise these here and also check further on (line 184) and change NODETYPE to match cluster system
## MODTRINITY will also be different on different clusters - it sets up the paths to the required executables
##################################################################################################################################
if [[ "$PBSTYPE" = "--pbspro" ]] ; then
JOBARRAY="-J"
JOBARRAY_ID="\$PBS_ARRAY_INDEX"
AFTEROKARRAY="afterok"
elif [[ "$PBSTYPE" = "--pbs" ]] ; then
## PBS torque:
JOBARRAY="-t"
JOBARRAY_ID="\$PBS_ARRAYID"
AFTEROKARRAY="afterokarray"
else # no paramaters present
F_USAGE
exit 0
fi
##############################################################################################################################################
########## Part 1: Set up file names for input directory and for output data dir and Trinity.pl command line ####################
##############################################################################################################################################
echo ""
## Ensure JOBPREFIX is not greatr than 11 characters as PBS-pro can not handle > 15 characters for total job name length
echo "submitting trinity jobs with prefix: "
echo " $JOBPREFIX"
###### Set input data directory - $DATADIR is CSIRO specific
echo "Input directory: "
echo " $DATADIRECTORY"
###### Set output data directory (OUTPUTDIR)
echo "Output directory: Scripts and output data will be written to:"
echo " $OUTPUTDIR"
mkdir -p "$OUTPUTDIR"
cd "$OUTPUTDIR"
### Modify STANDARD_JOB_DETAILS for analysis specific input to Trinity.pl
echo "The following trinity command line will be run:"
echo "$STANDARD_JOB_DETAILS"
echo ""
if [[ "$1" = "--pbs" ]] ; then
echo " Use: \"pbs_check.pl -t PBS_JOBID\" "
echo " To view stdout and stderr from each separate job while they are running"
fi
###########################################################################################################################################################
### Do some checking that files exist etc. (User should not modify)
### This sets the $DS variable
SET_DS "$FILENAMEINPUT"
###########################################################################################################################################################
################################# ###################################
################################# Part 2: Create the shell scripts to be run via the PBS batch system ###################################
################################# Users should modify WALLTIME and MEM dependent upon dataset size ###################################
################################# and NCPU to appropriate value for compute node cpu resources ###################################
################################# Check: "Trinity RNA-seq Assembler Performance Optimisation" (Henschel 2012) ###################################
################################# for current best practice. ###################################
################################# N.B. On busy clusters it may be best not to try to request ###################################
################################# a full nodes resources. i.e. If 8 cores per node are present, ###################################
################################# only request half of these. ###################################
###########################################################################################################################################################
###########################################################################################################################################################
############################## Script 1: Write script to run Inchworm ############################################
############################# MEM should equal JFMEM, which is the amount of memory requested for Jellyfish ############################################
JOBNAME1="$JOBPREFIX"_p1
NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P1" "$NCPU_P1" "$PBSNODETYPE_P1" "$WALLTIME_P1" "$JOBNAME1" "$ACCOUNT" "$PBSUSER" "$MODTRINITY" "$JOBPREFIX")
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p1"
#############################################################################################################################################################
############################## Script 2: Chrysalis::GraphFromFasta #############################################
############################## This script has a dependency on part 1 completion without error. #############################################
# It would be good to force an exit(0) before ReadsToTranscripts after checkpoint file /chrysalis/GraphFromIwormFasta.finished is
# written (i.e. add --no_run_readstotrans to Trinity and pass through to Chrysalis) as the script 3 can be started directly after.
# This may be less of an issue with the new (fast) version of Trinity::GraphFromFasta (since version 2012-06-08).
JOBNAME2="$JOBPREFIX"_p2
NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P2" "$NCPU_P2" "$PBSNODETYPE_P2" "$WALLTIME_P2" "$JOBNAME2" "$ACCOUNT" "$PBSUSER" "$MODTRINITY" "$JOBPREFIX")
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p2"
###########################################################################################################################################################
############################## Script 3: Script to run Chrysalis::ReadsToTranscripts ###################
############################## ReadsToTranscripts can be slow due to reads from disk, ###################
############################## This script has a dependency on part 2 completion with error. ###################
############################## Section script is skipped if enough time was given in Part 2. ###################
JOBNAME3="$JOBPREFIX"_p3
NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P3" "$NCPU_P3" "$PBSNODETYPE_P3" "$WALLTIME_P3" "$JOBNAME3" "$ACCOUNT" "$PBSUSER" "$MODTRINITY" "$JOBPREFIX")
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p3"
##########################################################################################################################################################
############################## Script 4a: Write script to call the Chrysalis QuantifyGraph array job ##################
############################## N.B. SLOTLIMIT="%x" indicates 'slot limit' i.e. the number of concurrent jobs to execute in an array ##################
############################## (Not available in PBSpro ) ##################
#### NB Disabling emails for arrays
#SLOTLIMIT="%64" # available for PBS Torque
JOBNAME4="$JOBPREFIX"_p4a
NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" 1gb 1 "$PBSNODETYPE_P4" 00:30:00 "$JOBNAME4" "$ACCOUNT" "$PBSUSER" "$MODTRINITY" "$JOBPREFIX" )
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p4a"
###########################################################################################################################################################
############################## Script Array part 4b: Write script to be run as an array Job . Runs Chrysalis::QuantifyGraph ##################
############################## This scipt has a dependency on part 4a being run. If an array component fails it will email user. ##################
############################## Files named quantifyGraph_commands_X are written with subset of total commands (X is the array ID from the PBS system) ##
JOBNAME4B="$JOBPREFIX"_p4b
NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P4" "$NCPU_P4" "$PBSNODETYPE_P4" "$WALLTIME_P4" "$JOBNAME4B" "$ACCOUNT" " " "$MODTRINITY" "$JOBPREFIX")
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p4b"
###########################################################################################################################################################
############################## Script Array 5b: Write script to run Butterfly Array Job ##################
############################## This script has a dependency on part 4a 4b being run. ##################
############################## Files named butterfly_commands_X are written with subset of total commands (X is the array ID from the PBS system) ###
JOBNAME5B="$JOBPREFIX"_p5b
NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P5" "$NCPU_P5" "$PBSNODETYPE_P5" "$WALLTIME_P5" "$JOBNAME5B" "$ACCOUNT" " " "$MODTRINITY" "$JOBPREFIX")
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p5b"
############################################################################################################################################################
############### ###################
############### Part 3: Write main control script that executes scripts that were created above. ###################
############### ###################
############################################################################################################################################################
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.cont"
############### Run the script written in Part 3
############### Checks to see what is current stage of calculation and executes scripts created in above code ###################
bash ""$JOBPREFIX"_run.sh"
exit 0