20251125
This commit is contained in:
@@ -0,0 +1,291 @@
|
||||
#!/usr/bin/env perl
|
||||
|
||||
=pod
|
||||
|
||||
=head1 NAME
|
||||
|
||||
=head1 USAGE
|
||||
|
||||
-in input file in FASTA/Q or posmap length file (e.g. posmap.scflen)
|
||||
-genome genome size in bp for estimating N lengths and indexes
|
||||
-single FASTA/Q has sequence in a single line (faster)
|
||||
-overwrite => Force overwrite
|
||||
-reads => Force processing as read data (no N50 statistics)
|
||||
-noreads => Force as not being read data. Good for cDNA assemblies with short contigs
|
||||
|
||||
=head1 AUTHORS
|
||||
|
||||
Alexie Papanicolaou 1
|
||||
|
||||
Ecosystem Sciences, CSIRO, Black Mountain Labs, Clunies Ross Str, Canberra, Australia
|
||||
alexie@butterflybase.org
|
||||
|
||||
=head1 DISCLAIMER & LICENSE
|
||||
|
||||
This software is released under the GNU General Public License version 3 (GPLv3).
|
||||
It is provided "as is" without warranty of any kind.
|
||||
You can find the terms and conditions at http://www.opensource.org/licenses/gpl-3.0.html.
|
||||
Please note that incorporating the whole software or parts of its code in proprietary software
|
||||
is prohibited under the current license.
|
||||
|
||||
=head1 BUGS & LIMITATIONS
|
||||
|
||||
None known so far.
|
||||
|
||||
=cut
|
||||
|
||||
use strict;
|
||||
use warnings;
|
||||
use Getopt::Long;
|
||||
use Pod::Usage;
|
||||
use Statistics::Descriptive;
|
||||
use Bio::SeqIO;
|
||||
$|=1;
|
||||
|
||||
my (@infiles,$user_genome_size,$is_fasta,$is_fastq,$is_single,$overwrite,$is_reads,$isnot_reads);
|
||||
GetOptions(
|
||||
'in=s{,}' => \@infiles,
|
||||
'single' =>\$is_single,
|
||||
'genome:s' => \$user_genome_size,
|
||||
'overwrite' => \$overwrite,
|
||||
'reads' =>\$is_reads,
|
||||
'noreads' =>\$isnot_reads,
|
||||
);
|
||||
if (!@infiles){
|
||||
@infiles = @ARGV;
|
||||
}
|
||||
pod2usage "No input files!\n" if !@infiles;
|
||||
die "Cannot ask for both reads and noreads options at the same time!\n" if $is_reads && $isnot_reads;
|
||||
|
||||
if ($is_reads && !$isnot_reads){
|
||||
print "Processing all data as reads\n";
|
||||
}
|
||||
if ($user_genome_size && $user_genome_size=~/\D/){
|
||||
if ($user_genome_size=~/^(\d+)kb$/i){
|
||||
$user_genome_size=int($1.'000');
|
||||
}
|
||||
elsif ($user_genome_size=~/^(\d+)mb$/i){
|
||||
$user_genome_size=int($1.'000000');
|
||||
}
|
||||
elsif ($user_genome_size=~/^(\d+)gb$/i){
|
||||
$user_genome_size=int($1.'000000000');
|
||||
}
|
||||
print "Genome set to ".&thousands($user_genome_size)." b.p.\n";
|
||||
}
|
||||
|
||||
foreach my $infile (@infiles){
|
||||
unless ($infile && -s $infile){warn("I need a posmap length file, e.g. .posmap.scflen for scaffolds\n");pod2usage;}
|
||||
my $outfile=$infile.'.n50';
|
||||
$outfile.='g' if ($user_genome_size);
|
||||
warn ("Outfile $outfile already exists\n") if -s $outfile && !$overwrite;
|
||||
next if -s $outfile && !$overwrite;
|
||||
my $total=int(0);
|
||||
my $gaps = int(0);
|
||||
my $seq_ref;
|
||||
my @head=`head $infile`;
|
||||
foreach (@head){
|
||||
if ($_=~/^>\S/){
|
||||
$is_fasta=1;
|
||||
print "FASTA file found!\n";
|
||||
last;
|
||||
}elsif($_=~/^@\S/){
|
||||
$is_fastq=1;
|
||||
print "FASTQ file found!\n";
|
||||
last;
|
||||
}
|
||||
}
|
||||
|
||||
print "Parsing file $infile...\n";
|
||||
if ($is_fasta){
|
||||
($total,$gaps,$seq_ref) = &process_fasta($infile);
|
||||
}
|
||||
elsif($is_fastq){
|
||||
($total,$gaps,$seq_ref) = &process_fastq($infile);
|
||||
}
|
||||
else {
|
||||
($total,$gaps,$seq_ref) = &process_csv($infile);
|
||||
}
|
||||
print "Preparing stats...\n";
|
||||
my ($mean,$n50,$n10,$n25,$n50_length,$n10_length,$n25_length,$scaffolds,$scaffolds_size,$smallest,$largest,$sequence_number,$sum,$genome_size) = &process_stats($seq_ref,$total);
|
||||
|
||||
if ($mean){
|
||||
open (OUT,">".$outfile);
|
||||
my $stat = Statistics::Descriptive::Full->new();
|
||||
$stat->add_data($seq_ref);
|
||||
#my $skew='';sprintf("%.2f",$stat->skewness());
|
||||
my $mean = sprintf("%.2f",$mean);
|
||||
my $median = $stat->median();
|
||||
my $var = sprintf("%.2f",$stat->variance());
|
||||
my $sd = sprintf("%.2f",$stat->standard_deviation());
|
||||
if (!$scaffolds || $scaffolds == 0){
|
||||
$scaffolds=$sequence_number;
|
||||
$scaffolds_size=$smallest;
|
||||
}
|
||||
print OUT "File: $infile\n";
|
||||
print OUT "TOTAL: ".&thousands($total)." bp in ".&thousands($sequence_number)." sequences\n";
|
||||
print OUT "\tof which ".&thousands($gaps)." are Ns/gaps.\n";
|
||||
print OUT "Mean: ".&thousands($mean)."\nStdev: ".&thousands($sd)."\n";
|
||||
print OUT "Median: ".&thousands($median)."\n";
|
||||
print OUT "Smallest: ".&thousands($smallest)."\nLargest: ".&thousands($largest)."\n";
|
||||
if (($mean >=1000 && !$is_reads) || $isnot_reads){
|
||||
print OUT "N10 length: ".&thousands($n10_length)."\nN10 Number: ".&thousands($n10)."\n";
|
||||
print OUT "N25 length: ".&thousands($n25_length)."\nN25 Number: ".&thousands($n25)."\n";
|
||||
print OUT "N50 length: ".&thousands($n50_length)."\nN50 Number: ".&thousands($n50)."\n";
|
||||
print OUT "Assuming a genome size of "
|
||||
.&thousands($user_genome_size)
|
||||
." then the top ".&thousands($scaffolds)
|
||||
." account for it (min "
|
||||
.&thousands($scaffolds_size)
|
||||
." bp)\n" if $user_genome_size;
|
||||
}else{
|
||||
print OUT "Reads found! Read coverage estimated to ".sprintf("%.2f",$total/$user_genome_size)."x using user provided genome size of ".&thousands($user_genome_size)."\n" if $user_genome_size;
|
||||
}
|
||||
close (OUT);
|
||||
print "Done, see $outfile\n";
|
||||
system("cat $outfile");
|
||||
}else {
|
||||
open (OUT,">".$outfile);
|
||||
print OUT "File: $infile\n";
|
||||
print OUT "TOTAL: $total bp in $sequence_number sequences\n";
|
||||
close (OUT);
|
||||
warn "Non fatal warning: Something went wrong in estimating the statistics. Maybe the provided genome length is much larger than sequence length or maybe less than 3 sequences provided?\n";
|
||||
}
|
||||
}
|
||||
########################################################################
|
||||
sub process_fasta(){
|
||||
print "Processing as FASTA\n";
|
||||
my $infile=shift;
|
||||
my @array ;
|
||||
my $total=int(0);
|
||||
my $gaps=int(0);
|
||||
my $counter = int(0);
|
||||
|
||||
if ($is_single){
|
||||
open (IN,$infile)||die;
|
||||
while (my $seq_id=<IN>) {
|
||||
my $seq=<IN>;
|
||||
my $length=length($seq)-1; # newline
|
||||
$counter+=length($seq_id)+$length+1;
|
||||
next unless $length;
|
||||
$gaps+=($seq=~tr/[N\-]//);
|
||||
push(@array,$length);
|
||||
$total+=$length;
|
||||
}
|
||||
close IN;
|
||||
}else{
|
||||
my $filein = new Bio::SeqIO(-file=>$infile , -format=>'fasta');
|
||||
while (my $seq_obj=$filein->next_seq()) {
|
||||
$counter+=length($seq_obj->seq().$seq_obj->description().' '.$seq_obj->id()) if $seq_obj->seq();
|
||||
my $length=$seq_obj->length();
|
||||
my $seq=$seq_obj->seq();
|
||||
$gaps+=($seq=~tr/[N\-]//);
|
||||
next unless $length;
|
||||
push(@array,$length);
|
||||
$total+=$length;
|
||||
}
|
||||
}
|
||||
print "\n";
|
||||
die "No data found or wrong format\n" unless $total;
|
||||
return ($total,$gaps,\@array);
|
||||
}
|
||||
sub process_fastq(){
|
||||
print "Processing as FASTQ\n";
|
||||
my $infile=shift;
|
||||
my @array ;
|
||||
my $total=int(0);
|
||||
my $gaps=int(0);
|
||||
my $counter = int(0);
|
||||
open (IN,$infile);
|
||||
while (my $seq_id=<IN>) {
|
||||
my $seq=<IN>;
|
||||
my $scrap=<IN>.<IN>;
|
||||
my $length=length($seq)-1; #newline
|
||||
$counter+=length($seq_id)+$length+1;
|
||||
next unless $length;
|
||||
$gaps+=($seq=~tr/[N\-]//);
|
||||
push(@array,$length);
|
||||
$total+=$length;
|
||||
}
|
||||
close IN;
|
||||
print "\n";
|
||||
die "No data found or wrong format\n" unless $total;
|
||||
return ($total,$gaps,\@array);
|
||||
}
|
||||
sub process_csv(){
|
||||
print "Processing as CSV\n";
|
||||
my $infile=shift;
|
||||
my @array ;
|
||||
my $total=int(0);
|
||||
my $gaps='N/A';
|
||||
my $counter = int(0);
|
||||
open (IN,$infile)||die ("Cannot open $infile\n");
|
||||
while (my $ln=<IN>){
|
||||
$counter+=length($ln);
|
||||
$ln=~/(\d+)$/;
|
||||
next unless $1;
|
||||
my $length= $1;
|
||||
push(@array,$1);
|
||||
$total+=$length;
|
||||
}
|
||||
close IN;
|
||||
die "No data found or wrong format\n" unless $total;
|
||||
return ($total,$gaps,\@array);
|
||||
}
|
||||
|
||||
sub process_stats(){
|
||||
my $sequences_ref = shift;
|
||||
my $total = shift;
|
||||
my $genome_size = int(0);
|
||||
my $mean = $total / scalar(@$sequences_ref);
|
||||
my ($n50,$n10,$n25,$n50_length,$n10_length,$n25_length,$scaffolds,$scaffolds_size,$smallest,$largest,$sequence_number,$sum);
|
||||
print "Sorting...";
|
||||
my @sequences=sort{$b<=>$a} @$sequences_ref;
|
||||
$smallest=$sequences[-1];
|
||||
$largest=$sequences[0];
|
||||
print " done!\n";
|
||||
$|=0;
|
||||
if (($mean < 1000 && !$isnot_reads) || $is_reads ){
|
||||
print "Reads detected. Ignoring N* calculations.\n";
|
||||
$genome_size=$total;
|
||||
$sequence_number = scalar(@$sequences_ref);
|
||||
return ($mean,$n50,$n10,$n25,$n50_length,$n10_length,$n25_length,$scaffolds,$scaffolds_size,$smallest,$largest,$sequence_number,$sum) if $mean <1000;
|
||||
}
|
||||
elsif (!$user_genome_size){
|
||||
print "Setting genome size for N* calculations to total consensus $total\n";
|
||||
$genome_size=$total;
|
||||
}elsif($user_genome_size){
|
||||
print "Setting genome size for N* calculations to user defined $user_genome_size\n";
|
||||
$genome_size = $user_genome_size;
|
||||
}
|
||||
|
||||
foreach my $sequence_length ( @sequences){
|
||||
$sum+=$sequence_length;
|
||||
$sequence_number++;
|
||||
if($sum >= $genome_size*0.1 && !$n10){
|
||||
$n10=$sequence_number;
|
||||
$n10_length=$sequence_length;
|
||||
}
|
||||
elsif($sum >= $genome_size*0.25 && !$n25){
|
||||
$n25=$sequence_number;
|
||||
$n25_length=$sequence_length;
|
||||
}
|
||||
elsif($sum >= $genome_size*0.5 && !$n50){
|
||||
$n50 = $sequence_number;
|
||||
$n50_length=$sequence_length;
|
||||
}elsif ($sum >= $genome_size && !$scaffolds){
|
||||
$scaffolds = $sequence_number;
|
||||
$scaffolds_size = $sequence_length;
|
||||
}
|
||||
|
||||
}
|
||||
print "Processed $sequence_number sequences\n";
|
||||
return ($mean,$n50,$n10,$n25,$n50_length,$n10_length,$n25_length,$scaffolds,$scaffolds_size,$smallest,$largest,$sequence_number,$sum,$genome_size);
|
||||
}
|
||||
|
||||
sub thousands($){
|
||||
my $val = shift;
|
||||
return int(0) if !$val;
|
||||
$val = sprintf("%.0f", $val);
|
||||
1 while $val =~ s/(.*\d)(\d\d\d)/$1,$2/;
|
||||
return $val;
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
################################################################################################################
|
||||
######################## README file ########################
|
||||
######################## Trinity PBS job submission with multi part dependencies ########################
|
||||
######################## Author: Josh Bowden, Alexie Papanicolaou, CSIRO ########################
|
||||
######################## Email: alexie@butterflybase.org ########################
|
||||
######################## Version 1.0 ########################
|
||||
################################################################################################################
|
||||
|
||||
DESCRIPTION: BASH shell scripts for submission of Trinity jobs to clusters that use PBS Torque or PBS Pro.
|
||||
The set of scripts stages the parts of the Trinity workflow into 6 stages:
|
||||
1/ Inchworm
|
||||
2/ Chrysalis::GraphFromFasta and Chrysalis::ReadsToTranscripts if walltime permits
|
||||
3/ Chrysalis::ReadsToTranscripts
|
||||
4/ Chrysalis::QuantifyGraph
|
||||
5/ Butterfly
|
||||
6/ Gather together resulting transcripts into "Trinity.fasta" file
|
||||
|
||||
Each stage is submitted as a PBS job, with dependencies i.e. the following stage will only execute after successfull completion of the stage before.
|
||||
Due to their parallel nature, stages 4 and 5 are submitted as 'array jobs', with each job made up of a user defined number of subtasks.
|
||||
|
||||
ADMINISTRATION setup instructions:
|
||||
trinity_pbs script install instructions:
|
||||
1. Copy all trinity_pbs.* files into a directory (we will call it "TRINITY_PBS_DIR") in a user accessible, read only, area.
|
||||
2. Add TRINITY_PBS_DIR to the PATH i.e. export or set PATH=TRINITY_PBS_DIR:$PATH
|
||||
3. Make sure trinity_pbs.sh is executable (chmod 755 trinity_pbs.sh)
|
||||
In the file trinity_pbs.sh, do the following:
|
||||
4. Change the "TRINITYPATH" variable to point to the trinity.pl installation directory.
|
||||
5. Set MEMDIRIN to the name of a node-local filesystem so a network drive is not needed for final parallel stages
|
||||
6. Set MODTRINITY so that the trinity executables will be available - load approprite modules and set PATH.
|
||||
7. Set PBSTYPE to --pbspro or --pbs, dependent on the system present.
|
||||
That should be all that is needed from an admin perspective. These scripts have been tested on PBS Torque 3.0.6 and PBS Pro 11.0.2.
|
||||
There may be PBS system specific changes due to PBS version incompatabilities.
|
||||
8. Maybe make any system-specific changes to the user-specific text README below or TRINITY.CONFIG.template before distributing
|
||||
|
||||
USER setup instructions:
|
||||
1. Users should make a copy of TRINITY.CONFIG.template (possibly a copy for each job they want to run) and then modify variables in it
|
||||
to suit the system the job is being run on (number of CPUs, amount of memory) and the expected runtime of the Trinity process
|
||||
(which is a function of the datset size). Further instructions are provided within the TRINITY.CONFIG.template file.
|
||||
2. While it is running, you can use qsub -u $USER to see your jobs
|
||||
3. We provide a script, pbs_check.pl, that allows you to see the progress of your jobs; just give the job id
|
||||
4. If any step fails, then some jobs that depended on a /successful/ completion of that step will be stuck in a HOLD (H) status
|
||||
5. The script trinity_kill.pl can be used to kill all running, queued or held jobs by passing it the output data directory (OUTPUTDIR).
|
||||
6. When you re-submit a job with trinity_pbs.sh <config file>, trinity_kill.pl will automatically run and stop and jobs in the directory specified by
|
||||
your config file
|
||||
(we take no responsibility if you delete the chrysalis directory and Trinity.fasta does not have all the data - e.g. because the PBS crashed)
|
||||
|
||||
|
||||
USER recommendations
|
||||
* Walltimes depend on how much data you have and also on your local HPC environment well (e.g. speed of I/O - hard disks and CPU configuration).
|
||||
In the beginning you may want to err towards higher walltimes, ask colleagues using the same machines. The default values worked well for us.
|
||||
* For quantify graph/Butterfly (steps 4/5), we start with a particular walltime (say 2h). Often not all jobs complete. Because it is a batch of many commands,
|
||||
the best approach is to simply re-launch trinity_pbs and it will continue to process the rest of the commands. Check the logfile output from the PBS
|
||||
to see if the job fails because one particular quantifygraph or Butterfly job takes longer than the given walltime (in that case: increase the walltime, run the command
|
||||
manually or decrease the number of reads used - -max_reads for quantifygraph). It is possible that it is a very long gene, a bacterial contaminant or
|
||||
some other oddity that is delaying your assembly.
|
||||
* Do not overload the I/O of the system (steps 4/5) by starting a lot of jobs (i.e. multiple trinity assemblies). In that case, increase the
|
||||
number of commands run within each step 4/5 batch (NUMPERARRAYITEM variable in the CONFIG)
|
||||
* NB: Always count how many sequences you get at the end (Trinity.fasta) to make sure that the script has completed as you expected.
|
||||
@@ -0,0 +1,122 @@
|
||||
##################################################################################################################################
|
||||
########################## ########################################
|
||||
########################## Trinity PBS job submission with multi part dependencies ########################################
|
||||
########################## User modifiyable input file ########################################
|
||||
########################## ########################################
|
||||
##################################################################################################################################
|
||||
### Author: Josh Bowden, CSIRO IM&T, Alexie Papanicolaou CSIRO CES
|
||||
### Email: alexie@butterflybase.org
|
||||
### Version 1.0
|
||||
###
|
||||
### Configuration file for script to split the Trinity workflow into multiple stages so as to efficiently request
|
||||
### and use appropriate resources (walltime and number of cores) on a computer cluster / supercomputer.
|
||||
###
|
||||
### User must set all the variables in this file to appropriate values
|
||||
### and then run the trinity_pbs.sh script with this file as input as shown below:
|
||||
###
|
||||
### Command line usage:
|
||||
### To start (or re-start) an analysis:
|
||||
### >trinity_pbs.sh TRINITY.CONFIG.template
|
||||
### To stop previously started PBS jobs on the queue:
|
||||
### >trinity_pbs.sh --rm OUTPUTDIR
|
||||
### Where:
|
||||
### TRINITY.CONFIG.template = user specific job details (i.e. the current file)
|
||||
### OUTPUTDIR = is path to output data directory (set below)
|
||||
###
|
||||
### If any stage fails, the jobs may be resubmitted and only the scripts that have not completed
|
||||
### will be resubmitted to the batch system. Either the scripts can be re-run (by using trinity_pbs.sh)
|
||||
### with original (or new) inputs from the current file (changing the variables below)
|
||||
### or the original scripts created by trinity_pbs.sh can be re-run by finding them in the output directory.
|
||||
###
|
||||
### Each stage and each array job will have a PBS output file sent to the output directory when the job finishes.
|
||||
### This means there will be many output files from the PBS system when the array job runs (from part 4 and 5 mostly).
|
||||
### If any part fails, errors will be specified in these output files.
|
||||
###
|
||||
### N.B. The trinity_pbs.sh file must have a number of system specific variables set by a system administrator
|
||||
###
|
||||
##################################################################################################################################
|
||||
|
||||
# USER must edit these:
|
||||
###### Set an email to which job progress and status will be sent to.
|
||||
UEMAIL=
|
||||
###### Set a valid account (if available), otherwise leave blank.
|
||||
ACCOUNT="#PBS -A sf-CSIRO"
|
||||
###### Select a value for JOBPRFIX that is not longer than 7 characters. NO spaces or other non-alphanumeric characters
|
||||
JOBPREFIX=
|
||||
|
||||
###### Set output data directory (OUTPUTDIR)
|
||||
###### OUTPUTDIR is where PBS scripts will be written and also Trinity results will be stored
|
||||
###### This area requires a large amount of space (possibly 100's of GB) and a high file count
|
||||
###### ($WORKDIR is a standard area on some systems, however users should check that it is valid on their machine)
|
||||
OUTPUTDIR="$WORKDIR"/trinityrnaseq/"$JOBPREFIX"
|
||||
|
||||
###### Set input data directory. This has to be explicitly set as it is used in other internal scripts.
|
||||
###### Make sure you include the final forward slash. Defaults to current directory
|
||||
DATADIRECTORY=$PWD/
|
||||
###### Set input filenames, you can use wildcards if you embed the filename is 'single quotes'
|
||||
FILENAMELEFT='*_left.fasta' # change this
|
||||
FILENAMERIGHT='*_right.fasta' #change this
|
||||
FILENAMESINGLE=single.fasta # change this or set it to empty
|
||||
SEQTYPE=fa # change this to fa (FASTA) fq (FASTQ) or cfa/cfq (FASTA or FASTQ colour space SOLiD ABI)
|
||||
|
||||
|
||||
# User may opt to change these:
|
||||
# we set --max_reads_per_graph to 1million because very high I/O is needed otherwise. It is unlikely that a transcript needs more than 1 million reads to be assembled....
|
||||
FILENAMEINPUT=" --seqType "$SEQTYPE" --left "$DATADIRECTORY""$FILENAMELEFT" --right "$DATADIRECTORY""$FILENAMERIGHT" --max_reads_per_graph 1000000 "
|
||||
### STANDARD_JOB_DETAILS sets analysis specific input to Trinity.pl
|
||||
### N.B. do not use --CPU or JM flag as this is automatically appended
|
||||
STANDARD_JOB_DETAILS="Trinity.pl "$FILENAMEINPUT" --output "$OUTPUTDIR""
|
||||
|
||||
|
||||
# This is where you specify resource limits. We provide some defaults values
|
||||
# Ultimately settings depends on your data size and complexity
|
||||
# Steps that go beyond their walltime will not complete. Edit these values and resubmit
|
||||
### Stage P1: Time and resources required for Inchworm stage
|
||||
### Only use at maximum, half the available CPUs on a node
|
||||
# - Inchworm will not efficiently use any more than 4 CPUs and you will have to take longer for resources to be assigned
|
||||
WALLTIME_P1="2:00:00"
|
||||
MEM_P1="20gb" # will use it for --JM
|
||||
NCPU_P1="4"
|
||||
PBSNODETYPE_P1="any" # ask you system administrator what Nodetypes exists
|
||||
|
||||
### Stage P2: Time and resources required for Chrysalis stage
|
||||
### Starts with Bowtie alignment and post-processing of alignment file
|
||||
### All CPUs presenct can be used for the Chrysalis parts.
|
||||
#They may take a while to be provisioned, so the less request, possibly the faster the jobs turnaround.
|
||||
# For one step (the parallel sort) it needs as much memory as specified in P1. Less memory, means more I/O for sorting
|
||||
# increase for more lanes of data: 3 lanes-> 60gb of RAM and 24h of time will do it.
|
||||
# The bowtie step will take considerable amount of time with more data
|
||||
WALLTIME_P2="12:00:00"
|
||||
MEM_P2="20gb" # will use it for the parallel sort of the SAM after alignment
|
||||
NCPU_P2="6"
|
||||
PBSNODETYPE_P2="any"
|
||||
|
||||
### Stage P3: This is a backup stage for Chrysalis - only runs if time ran out in P2 above.
|
||||
### This will need about 1 day per lane
|
||||
WALLTIME_P3="18:00:00"
|
||||
MEM_P3="8gb"
|
||||
NCPU_P3="6"
|
||||
PBSNODETYPE_P3="medium"
|
||||
|
||||
### Stage P4: QuantifyGraph graph runs in many parallel parts
|
||||
### Tasks that fail will reamain in the OUTPUTDIR/chrysalis/quantifyGraph_commands.XYZ files (XYZ is a number)
|
||||
### The remaining tasks can be run by running the job submission command "trinity_pbs.sh <config_file>" again.
|
||||
NUMPERARRAYITEM_P4=5000
|
||||
WALLTIME_P4="00:30:00"
|
||||
MEM_P4="4gb"
|
||||
NCPU_P4="1"
|
||||
PBSNODETYPE_P4="medium"
|
||||
|
||||
### Stage P5: Butterfly options. Some butterfly jobs can take exceedingly long. Users may need to restart trinity_pbs multiple times to complete.
|
||||
### Tasks that fail will reamain in the OUTPUTDIR/chrysalis/butterfly_commands.adj.XYZ files (XYZ is a number)
|
||||
### Often there will be a few tasks that take a lot longer than others so multiple submissions to the cluster may be required.
|
||||
### The remaining tasks can be run by running the job submission command "trinity_pbs.sh <config_file>" again.
|
||||
### Running the long running tasks seperately may also be a good option.
|
||||
NUMPERARRAYITEM_P5=5000
|
||||
WALLTIME_P5="01:00:00"
|
||||
MEM_P5="10gb"
|
||||
NCPU_P5="1"
|
||||
PBSNODETYPE_P5="medium"
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
#!/usr/bin/env perl
|
||||
|
||||
=pod
|
||||
|
||||
=head1 NAME
|
||||
|
||||
Monitor a PBS job progress
|
||||
|
||||
=head1 USAGE
|
||||
|
||||
<job identifier> [options]
|
||||
|
||||
Options:
|
||||
one of
|
||||
-top use top on execution host
|
||||
-dump dump output or error (default)
|
||||
-follow follow output/error
|
||||
-tail tail of output/error (the last 10 lines)
|
||||
-head head of output/error (the first 10 lines)
|
||||
|
||||
and also
|
||||
-e|error Show stderr instead of stdout
|
||||
-s|spool Location of spool directory (defaults to /var/spool/PBS/spool)
|
||||
|
||||
=head1 LICENSE
|
||||
|
||||
Released under the MIT License - Alexie Papanicolaou 2012, CSIRO Ecosystem Sciences, alexie@butterflybase.org
|
||||
|
||||
=cut
|
||||
|
||||
|
||||
use strict;
|
||||
use warnings;
|
||||
use Pod::Usage;
|
||||
use Getopt::Long;
|
||||
my ($head,$follow,$tail,$dump,$show_error,$spool_dir,$top);
|
||||
GetOptions(
|
||||
'top' => \$top,
|
||||
'head' => \$head,
|
||||
'follow' => \$follow,
|
||||
'tail' => \$tail,
|
||||
'cat|dump' => \$dump,
|
||||
's|spool_dir:s' => \$spool_dir,
|
||||
'error' => \$show_error,
|
||||
);
|
||||
|
||||
my $method;
|
||||
if ($top){
|
||||
$method = 'top';
|
||||
}elsif ($head){
|
||||
$method = 'head';
|
||||
}elsif ($follow){
|
||||
$method = 'follow'
|
||||
}elsif ($tail){
|
||||
$method = 'tail';
|
||||
}else{
|
||||
$method = 'cat';
|
||||
}
|
||||
|
||||
my $jobid = shift;
|
||||
|
||||
pod2usage unless $jobid;
|
||||
|
||||
$spool_dir = $spool_dir ? $spool_dir : '/var/spool/PBS/spool'; # exists on exec host but necessarily on submit host
|
||||
my $user = $ENV{'USER'};
|
||||
my $exec = 'ssh ';
|
||||
pod2usage "No job ID provided\n" unless $jobid;
|
||||
$jobid=~/^(\w+\[?\d*\]?)/;
|
||||
$jobid=$1 || pod2usage "Not a valid job ID $jobid\n";
|
||||
|
||||
my $pbs_server = `qstat -Bf|grep ^Server`;
|
||||
chomp($pbs_server);
|
||||
$pbs_server=~s/^Server: //;
|
||||
die "No PBS server found. Is it online?\n" unless $pbs_server;
|
||||
|
||||
my $node=`qstat -f $jobid|grep exec_host`;
|
||||
$node =~/exec_host\s+=\s+(\w+)/;
|
||||
$node = $1 || "No valid host found. Is $jobid a live job?\n";
|
||||
my $cmd;
|
||||
|
||||
if ($method=~/^c/){
|
||||
$cmd = 'cat';
|
||||
}elsif ($method=~/^ta/){
|
||||
$cmd = 'tail';
|
||||
}elsif ($method=~/^f/){
|
||||
$cmd = 'tail -f';
|
||||
}elsif ($method=~/^h/){
|
||||
$cmd = 'head';
|
||||
}else {
|
||||
$cmd = $method;
|
||||
}
|
||||
if ($method eq 'top'){
|
||||
system("ssh $node -t top");
|
||||
}else{
|
||||
$exec.= " $node $cmd $spool_dir/$jobid.$pbs_server.OU" if !$show_error;
|
||||
$exec.= " $node $cmd $spool_dir/$jobid.$pbs_server.ER" if $show_error;
|
||||
system($exec);
|
||||
}
|
||||
print "\n#\tQPEEK complete for job $jobid. Host was: $node\n";
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
#!/usr/bin/env perl
|
||||
|
||||
use strict;
|
||||
use warnings;
|
||||
|
||||
my $me = $ENV{'USER'};
|
||||
my @jobs_running_ln = `qstat -u $ENV{'USER'}`;
|
||||
if (!@jobs_running_ln || scalar(@jobs_running_ln)<1){
|
||||
print "No running jobs for user $me\n";
|
||||
exit();
|
||||
}
|
||||
|
||||
my %jobs_running;
|
||||
foreach my $job_ln (@jobs_running_ln){
|
||||
$job_ln=~/^(\S+)/;
|
||||
$jobs_running{$1}=1 if $1;
|
||||
}
|
||||
|
||||
my $dir=$ARGV[0] ? $ARGV[0] : '.';
|
||||
if ($dir && -d $dir){
|
||||
# read jobnumbers in current directory (and directory passed as variable) and kill them (start at the last job and move to oldest)
|
||||
if (-s $dir."/jobnumbers.out"){
|
||||
my @job_sub = `tac $dir/jobnumbers.out`;
|
||||
chomp(@job_sub);
|
||||
foreach my $job (@job_sub){
|
||||
system("qdel -W force $job") if $jobs_running{$job};
|
||||
# twice to make sure
|
||||
system("qdel -W force $job >/dev/null 2>/dev/null") if $jobs_running{$job};
|
||||
}
|
||||
unlink($dir."/jobnumbers.out");
|
||||
}
|
||||
else{
|
||||
print "No previous jobs found\n";
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
#!/bin/bash
|
||||
|
||||
# much slower than perl version
|
||||
# read jobnumbers in current directory (and directory passed as variable) and kill them (start at the last job and move to oldest)
|
||||
|
||||
if [ $1 ]; then
|
||||
if [ -e "$1/jobnumbers.out" ];then
|
||||
tac "$1/jobnumbers.out" |
|
||||
while read line
|
||||
do
|
||||
echo "$1/jobnumbers.out": Stopping $line
|
||||
qdel -W force $line
|
||||
qdel -W force $line >/dev/null 2>/dev/null
|
||||
done
|
||||
rm -f "$1/jobnumbers.out"
|
||||
fi
|
||||
exit 0
|
||||
fi
|
||||
|
||||
if [ -e jobnumbers.out ];then
|
||||
tac jobnumbers.out |
|
||||
while read line
|
||||
do
|
||||
echo jobnumbers.out: Stopping $line
|
||||
qdel -W force $line
|
||||
qdel -W force $line >/dev/null 2>/dev/null
|
||||
done
|
||||
rm -f jobnumbers.out
|
||||
else
|
||||
echo "No previous jobs found"
|
||||
fi
|
||||
@@ -0,0 +1,70 @@
|
||||
##################################################################################################################################
|
||||
########################## ########################################
|
||||
########################## Trinity PBS job submission with multi part dependencies ########################################
|
||||
########################## ########################################
|
||||
##################################################################################################################################
|
||||
##################################################################################################################################
|
||||
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
|
||||
### Version 1.0
|
||||
########### The main control script, that we will run and can be run seperately later if the job fails at intermediate stages ###################
|
||||
############################################################################################################################################################
|
||||
|
||||
RUNVAR1=""$HASHBANG"
|
||||
|
||||
echo Killing any running jobs in "$OUTPUTDIR"
|
||||
trinity_kill.pl "$OUTPUTDIR" 2> /dev/null >/dev/null
|
||||
echo Checking and submitting any jobs
|
||||
###########################################################################################################################################################
|
||||
################ Use the Trinity/Chrysalis checkpoint files to work out what part of job remains #########################################
|
||||
################ and queue only those parts that are still required #########################################
|
||||
|
||||
if [ -s \""$OUTPUTDIR"/Trinity.fasta\" ] ; then
|
||||
echo 'Trinity seemingly finished: Trinity.fasta present in output directory. If you think this is wrong, delete Trinity.fasta and re-submit'
|
||||
exit 0
|
||||
fi
|
||||
|
||||
|
||||
if [ ! -e \""$OUTPUTDIR"/inchworm.K25.L25"$DS"fa.finished\" ] ;then # do the whole analysis
|
||||
PBS_JOB1=\`qsub "$JOBNAME1".sh\`
|
||||
PBS_JOB2=\`qsub -W depend=afterok:\$PBS_JOB1 "$JOBNAME2".sh\`
|
||||
PBS_JOB3=\`qsub -W depend=afternotok:\$PBS_JOB2 "$JOBNAME3".sh\`
|
||||
PBS_JOB4_2=\`qsub -W depend=afterok:\$PBS_JOB2,afterany:\$PBS_JOB3 "$JOBNAME4".sh\`
|
||||
echo "$JOBNAME1".sh submitted ; echo \"\$PBS_JOB1\" > jobnumbers.out ;
|
||||
echo "$JOBNAME2".sh submitted ; echo \"\$PBS_JOB2\" >> jobnumbers.out ;
|
||||
echo "$JOBNAME3".sh submitted ; echo \"\$PBS_JOB3\" >> jobnumbers.out ;
|
||||
echo "$JOBNAME4".sh post-job "$JOBNAME2" submitted ; echo \"\$PBS_JOB4_2\" >> jobnumbers.out ;
|
||||
else # do analysis after inchworm only
|
||||
if [ ! -e \""$OUTPUTDIR"/chrysalis/GraphFromIwormFasta.finished\" ] ; then # start analysis after inchworm
|
||||
PBS_JOB2=\`qsub "$JOBNAME2".sh\`
|
||||
PBS_JOB3=\`qsub -W depend=afternotok:\$PBS_JOB2 "$JOBNAME3".sh\`
|
||||
PBS_JOB4_2=\`qsub -W depend=afterok:\$PBS_JOB2,afterany:\$PBS_JOB3 "$JOBNAME4".sh\`
|
||||
echo "$JOBNAME2".sh submitted ; echo \"\$PBS_JOB2\" > jobnumbers.out ;
|
||||
echo "$JOBNAME3".sh submitted ; echo \"\$PBS_JOB3\" >> jobnumbers.out ;
|
||||
echo "$JOBNAME4".sh post-job "$JOBNAME2" submitted ; echo \"\$PBS_JOB4_2\" >> jobnumbers.out ;
|
||||
else
|
||||
if [ ! -e \""$OUTPUTDIR"/chrysalis/readsToComponents.finished\" ] ;then # start analysis at Chrisyalis ReadsToTranscripts - which is slow due to I/O
|
||||
PBS_JOB3=\`qsub "$JOBNAME3".sh\`
|
||||
PBS_JOB4=\`qsub -W depend=afterok:\$PBS_JOB3 "$JOBNAME4".sh\`
|
||||
echo "$JOBNAME3".sh submitted ; echo \"\$PBS_JOB3\" > jobnumbers.out ;
|
||||
echo "$JOBNAME4".sh post-job "$JOBNAME3" submitted ; echo \"\$PBS_JOB4\" >> jobnumbers.out ;
|
||||
else # Run Chrysalis QuantifyGraph and then Butterfly
|
||||
PBS_JOB4=\`qsub "$JOBNAME4".sh\`
|
||||
echo "$JOBNAME4".sh submitted ; echo \"\$PBS_JOB4\" > jobnumbers.out ;
|
||||
fi
|
||||
fi
|
||||
echo When JOB ID \"\$PBS_JOB4\" finishes successfully - see output: \"$OUTPUTDIR\"/\"$JOBNAME4\".o\"\$PBS_JOB4\" - either re-run the submit command or use the following command to get all the data into a single file:
|
||||
echo ' find "$OUTPUTDIR"/chrysalis -name *allProbPaths.fasta -exec cat {} \\; > "$OUTPUTDIR"/Trinity.fasta '
|
||||
fi
|
||||
"
|
||||
|
||||
echo "${RUNVAR1}" | cat -> ""$JOBPREFIX"_run.sh"
|
||||
chmod 744 ""$JOBPREFIX"_run.sh"
|
||||
echo "To restart these jobs run either the same command again:"
|
||||
echo " trinity_pbs.sh <the same .config file> "
|
||||
echo " or the following script: "
|
||||
echo " "$JOBPREFIX"_run.sh found in the output directory "$OUTPUTDIR" "
|
||||
echo "To stop these jobs run:"
|
||||
echo " trinity_kill.pl "$OUTPUTDIR" "
|
||||
echo "To check progress of these jobs run:"
|
||||
echo " qstat -u "$USER""
|
||||
echo ""
|
||||
@@ -0,0 +1,179 @@
|
||||
##################################################################################################################################
|
||||
########################## ########################################
|
||||
########################## Trinity PBS job submission with multi part dependencies ########################################
|
||||
########################## ########################################
|
||||
##################################################################################################################################
|
||||
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
|
||||
### Version 1.0
|
||||
### Function definitions and PBS version specific information
|
||||
##################################################################################################################################
|
||||
|
||||
######################################################################################################################
|
||||
### Return usage information if user requests --help -h or another incorrect input flag.
|
||||
######################################################################################################################
|
||||
function F_USAGE {
|
||||
echo ""
|
||||
echo " Usage: "
|
||||
echo ""
|
||||
echo " To start an analysis: "
|
||||
echo " trinity_pbs.sh <config.file> "
|
||||
echo ""
|
||||
echo " To stop previously started PBS jobs in the queue: "
|
||||
echo " trinity_kill.pl OUTPUTDIR "
|
||||
echo ""
|
||||
echo " Where:"
|
||||
echo " <config.file> = contains specific job details (see below)"
|
||||
echo " OUTPUTDIR = is path to output data directory"
|
||||
echo ""
|
||||
echo " Note: Modify the values in CONFIG_FILE to suit compute cluster and "
|
||||
echo " input data size and nameing conventions:"
|
||||
echo ""
|
||||
echo " <config.file> contains data and job input details and user defined variables "
|
||||
echo " and user and cluster specific options:"
|
||||
echo " UEMAIL, DATADIRECTORY, OUTPUTDIR, JOBPREFIX, ACCOUNT "
|
||||
echo " STANDARD_JOB_DETAILS"
|
||||
echo " per job details:"
|
||||
echo " NCPU, WALLTIME, MEM, MEMDIR, NUMPERARRAYITEM"
|
||||
echo ""
|
||||
}
|
||||
|
||||
|
||||
######################################################################################################################
|
||||
### Function to return the PBS header information, depending on PBS type.
|
||||
######################################################################################################################
|
||||
# NODESCPUS=$(F_GETNODESTRING $1 $MEM $NCPU large $WALLTIME $JOBNAME $ACCOUNT $PBSUSER $MODTRINITY $JOBPREFIX)
|
||||
# NODESCPUS=$(F_GETNODESTRING $1 $2 $3 $4 $5 $6 $7 $8 $9 ${10} )
|
||||
function F_GETNODESTRING {
|
||||
if [[ $1 = "--pbspro" ]] ; then
|
||||
echo "
|
||||
#PBS -l select=1:ncpus="$3":NodeType="$4":mem="$2"
|
||||
#PBS -l walltime="$5"
|
||||
#PBS -N "$6"
|
||||
"$7"
|
||||
#PBS -j oe
|
||||
#PBS -m a
|
||||
#PBS -V
|
||||
"$8"
|
||||
"$9"
|
||||
"
|
||||
elif [[ $1 = "--pbs" ]] ; then
|
||||
echo "
|
||||
#PBS -l nodes=1:ppn="$3"
|
||||
#PBS -l vmem="$2"
|
||||
#PBS -l walltime="$5"
|
||||
#PBS -N "$6"
|
||||
#PBS -j oe
|
||||
#PBS -m a
|
||||
"$8"
|
||||
#PBS -V
|
||||
"$9"
|
||||
"
|
||||
fi
|
||||
echo " JOBNAME=$6"
|
||||
echo " JOBPREFIX=${10}"
|
||||
}
|
||||
|
||||
|
||||
######################################################################################################################
|
||||
### Do some checking that files exist etc.
|
||||
######################################################################################################################
|
||||
#DS=$(F_GETNODESTRING $FILENAMEINPUT $DATADIRECTORY $FILENAMESINGLE $FILENAMELEFT $FILENAMERIGHT)
|
||||
# DS=$(F_GETNODESTRING $1 $2 $3 $4 $5 )
|
||||
function SET_DS {
|
||||
|
||||
# add the correct filename extension so we can check if inchworm has been run in part 3 below
|
||||
if [[ "$1" == *FR* ]] || [[ "$1" == *RF* ]] ;
|
||||
then
|
||||
DS="."
|
||||
else
|
||||
DS=".DS."
|
||||
fi
|
||||
|
||||
}
|
||||
|
||||
#AP: i removed this because a) it didn't stop the script from progressing when there was an error, b) it was not handling multiple input files
|
||||
function F_CHECKFILES {
|
||||
|
||||
# Check that input data file(s) exist.
|
||||
if [[ "$1" == *--single* ]] ; then
|
||||
if [ ! -f ""$2""$3"" ] ; then
|
||||
echo "Input file does not exist: "
|
||||
echo " "$2""$3""
|
||||
exit 1
|
||||
fi
|
||||
else # check that both left and right filenames exist
|
||||
if [ ! -f ""$2""$4"" ] ; then
|
||||
echo "Input file does not exist: "
|
||||
echo " "$2""$4""
|
||||
exit 1
|
||||
fi
|
||||
if [ ! -f ""$2""$5"" ] ; then
|
||||
echo "Input file does not exist: "
|
||||
echo " "$2""$5""
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
|
||||
######################################################################################################################
|
||||
### Checking that files containing stage specific script data exist
|
||||
######################################################################################################################
|
||||
#F_WRITESCRIPT( scriptname $SOURCENAME )
|
||||
#F_WRITESCRIPT( $1 $2 )
|
||||
function F_WRITESCRIPT {
|
||||
if [ -e "$2" ] ; then
|
||||
source "$2"
|
||||
else
|
||||
echo ""$1" requires file \""$2"\" to be present in the current directory"
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
|
||||
|
||||
##################################################################################################################################
|
||||
# This is used to stop all jobs sent to the PBS queue
|
||||
# requires path to the filename as second argument
|
||||
# The order of the following tests is important.
|
||||
##################################################################################################################################
|
||||
if [[ "$1" = "--rm" ]] || [[ "$1" = "-rm" ]] ; then
|
||||
if [[ -e $2/jobnumbers.out ]] ; then
|
||||
cat $2/jobnumbers.out | while read LINE; do
|
||||
qdel $LINE || { echo "continuing..." ;}
|
||||
done
|
||||
#echo " Any parallel "
|
||||
else
|
||||
echo " Could not stop jobs as could not open file: "
|
||||
echo " $2"jobnumbers.out""
|
||||
fi
|
||||
exit 0
|
||||
fi
|
||||
|
||||
|
||||
|
||||
if [[ "$1" = "--help" ]] || [[ "$1" = "-h" ]] || [[ "$1" = "-?" ]]; then
|
||||
F_USAGE
|
||||
exit 0
|
||||
fi
|
||||
|
||||
|
||||
##################################################################################################################################
|
||||
## User email information required from command line:
|
||||
## Provide a valid email address so PBS can email user on start and finish of execution of individual parts (not part 4b or 5b though)
|
||||
##################################################################################################################################
|
||||
|
||||
if [[ ! -e "$1" ]] ; then
|
||||
echo "input file does not exist: "$1""
|
||||
echo "Please provide an input file"
|
||||
F_USAGE
|
||||
exit 0
|
||||
fi
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
##################################################################################################################################
|
||||
########################## ########################################
|
||||
########################## Trinity PBS job submission with multi part dependencies ########################################
|
||||
########################## ########################################
|
||||
##################################################################################################################################
|
||||
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
|
||||
### Version 1.0
|
||||
### Inchworm P1 script
|
||||
##################################################################################################################################
|
||||
|
||||
if [[ $MEM_P1 =~ ^([0-9]+) ]]; then
|
||||
let MEM_BASE="${BASH_REMATCH[1]}"
|
||||
JM_MEM="$MEM_BASE"G
|
||||
else
|
||||
echo No memory given for kmer counter: "$MEM_P1"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
JOBSTRING1=""$HASHBANG"
|
||||
"$NODESCPUS"
|
||||
|
||||
cd "$OUTPUTDIR"
|
||||
export OMP_NUM_THREADS="$NCPU_P1"
|
||||
export KMP_AFFINITY=compact
|
||||
# this runs Inchworm only
|
||||
"$STANDARD_JOB_DETAILS" --JM "$JM_MEM" --CPU "$NCPU_P1" --no_run_chrysalis
|
||||
"
|
||||
# Write the JOBSTRING1 to a file for later execution
|
||||
echo "${JOBSTRING1}" | cat -> ""$JOBNAME1".sh"
|
||||
@@ -0,0 +1,43 @@
|
||||
##################################################################################################################################
|
||||
########################## ########################################
|
||||
########################## Trinity PBS job submission with multi part dependencies ########################################
|
||||
########################## ########################################
|
||||
##################################################################################################################################
|
||||
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
|
||||
### Version 1.0
|
||||
### Chrysalis P2 script
|
||||
##################################################################################################################################
|
||||
|
||||
if [[ $MEM_P2 =~ ^([0-9]+) ]]; then
|
||||
let MEM_BASE="${BASH_REMATCH[1]}"
|
||||
let ALIGN_MEM=$MEM_BASE-5
|
||||
if [ $ALIGN_MEM -le 0 ];then
|
||||
let ALIGN_MEM=$MEM_BASE-2
|
||||
if [ $ALIGN_MEM -le 0 ];then
|
||||
echo "Memory requested for MEM_P2 is too low. Ask for at least 5 gigabytes"
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
ALIGN_MEM="$ALIGN_MEM"G
|
||||
else
|
||||
echo No memory given: "$MEM_P2"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
|
||||
JOBSTRING2=""$HASHBANG"
|
||||
"$NODESCPUS"
|
||||
cd "$OUTPUTDIR"
|
||||
# set stack size to unlimited for Chrysalis (part 1)
|
||||
ulimit -s unlimited
|
||||
export OMP_NUM_THREADS="$NCPU_P2"
|
||||
export KMP_AFFINITY=scatter
|
||||
# this runs Chrysalis::GraphFromFasta and maybe Chrysalis::GraphFromFasta if there is still walltime
|
||||
"$STANDARD_JOB_DETAILS" --JM "$ALIGN_MEM" --CPU "$NCPU_P2" --no_run_quantifygraph
|
||||
"
|
||||
|
||||
|
||||
# Write the JOBSTRING2 to a file for later execution
|
||||
echo "${JOBSTRING2}" | cat -> ""$JOBNAME2".sh"
|
||||
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
##################################################################################################################################
|
||||
########################## ########################################
|
||||
########################## Trinity PBS job submission with multi part dependencies ########################################
|
||||
########################## ########################################
|
||||
##################################################################################################################################
|
||||
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
|
||||
### Version 1.0
|
||||
### Chrysalis P3 script (only run if Chrysalis P3 has not completed)
|
||||
##################################################################################################################################
|
||||
|
||||
if [[ $MEM_P3 =~ ^([0-9]+) ]]; then
|
||||
let MEM_BASE="${BASH_REMATCH[1]}"
|
||||
let ALIGN_MEM=$MEM_BASE-5
|
||||
if [ $ALIGN_MEM -le 0 ];then
|
||||
let ALIGN_MEM=$MEM_BASE-2
|
||||
if [ $ALIGN_MEM -le 0 ];then
|
||||
echo "Memory requested for MEM_P3 is too low. Ask for at least 5 gigabytes"
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
ALIGN_MEM="$ALIGN_MEM"G
|
||||
else
|
||||
echo No memory given: "$MEM_P3"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
JOBSTRING3=""$HASHBANG"
|
||||
"$NODESCPUS"
|
||||
cd "$OUTPUTDIR"
|
||||
ulimit -s unlimited
|
||||
export OMP_NUM_THREADS="$NCPU_P3"
|
||||
export KMP_AFFINITY=scatter
|
||||
# this runs Chrysalis::ReadsToTranscripts if it has not completed in the previous step
|
||||
"$STANDARD_JOB_DETAILS" --JM "$ALIGN_MEM" --CPU "$NCPU_P3" --no_run_quantifygraph
|
||||
"
|
||||
|
||||
# Write the JOBSTRING3 to a file for later execution
|
||||
echo "${JOBSTRING3}" | cat -> ""$JOBNAME3".sh"
|
||||
@@ -0,0 +1,156 @@
|
||||
##################################################################################################################################
|
||||
########################## ########################################
|
||||
########################## Trinity PBS job submission with multi part dependencies ########################################
|
||||
########################## ########################################
|
||||
##################################################################################################################################
|
||||
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
|
||||
### Version 1.0
|
||||
### QuantifyGraph and Butterfly p4a Script
|
||||
##################################################################################################################################
|
||||
# we will not use array in order to ensure only jobs that have not finished are Submitting.
|
||||
# this does cause a problem when wanting to kill them manually but best to use kill script
|
||||
## JOBPREFIX is passed via HASHBANG
|
||||
JOBSTRING4=""$HASHBANG"
|
||||
"$NODESCPUS"
|
||||
JOB_CHRYSALIS="$JOBPREFIX"_p4b
|
||||
JOB_BUTTERFLY="$JOBPREFIX"_p5b
|
||||
cd "$OUTPUTDIR"
|
||||
|
||||
rm -f \$JOB_CHRYSALIS.jobnames \$JOB_BUTTERFLY.jobnames
|
||||
FILENAME=""$OUTPUTDIR"/chrysalis/quantifyGraph_commands"
|
||||
FILENAMEBFLY=""$OUTPUTDIR"/chrysalis/butterfly_commands"
|
||||
|
||||
if [ ! -e \$FILENAME.pbs ];then
|
||||
sed -e 's/.*/if [[ -e SEDPLACEHOLDER ]]; then &;fi/' \$FILENAME|sed -r 's/(-e\s)SEDPLACEHOLDER(\s.+-i\s)(\S+).tmp/\1\3.tmp\2\3.tmp/' > \$FILENAME.pbs
|
||||
split -d -a 3 -l "1000" \$FILENAME.pbs \$FILENAME.pbs.
|
||||
sleep 5
|
||||
fi
|
||||
if [ ! -e \$FILENAMEBFLY.pbs ];then
|
||||
sed -e 's/.*/if [ -e SEDPLACEHOLDER ]; then &;fi/' \$FILENAMEBFLY|sed -r 's/(-e\s)SEDPLACEHOLDER(\s.+-C\s)(\S+)/\1\3.out\2\3/' > \$FILENAMEBFLY.pbs
|
||||
split -d -a 3 -l "1000" \$FILENAMEBFLY.pbs \$FILENAMEBFLY.pbs.
|
||||
sleep 5
|
||||
fi
|
||||
|
||||
FILENAME=\$FILENAME.pbs
|
||||
FILENAMEBFLY=\$FILENAMEBFLY.pbs
|
||||
|
||||
NUMCMDS=\`ls -l \$FILENAME.??? | wc -l\`
|
||||
NUMCMDSBFLY=\`ls -l \$FILENAMEBFLY.??? | wc -l\`
|
||||
let NUMCMDS=\$NUMCMDS-1
|
||||
let NUMCMDSBFLY=\$NUMCMDSBFLY-1
|
||||
let SUBMITTED_C=0
|
||||
let SUBMITTED_B=0
|
||||
|
||||
for ((JOBID=0;JOBID<=\$NUMCMDS;++JOBID));do
|
||||
JOB_INDEX_PADDED=\`printf "%03d" \$JOBID\`
|
||||
MYJOBQ=\""\$FILENAME".\$JOB_INDEX_PADDED\"
|
||||
MYJOBB=\""\$FILENAMEBFLY".\$JOB_INDEX_PADDED\"
|
||||
JOB_FILESIZE_Q=\$(stat -c%s \$MYJOBQ)
|
||||
JOB_FILESIZE_B=\$(stat -c%s \$MYJOBB)
|
||||
|
||||
#if some Q have completed:
|
||||
if [ -s \"\$MYJOBQ.completed\" ] ; then
|
||||
JOB_COMPLETED_FILESIZE_Q=\$(stat -c%s \"\$MYJOBQ.completed\")
|
||||
# if not all have completed then run both Q and B
|
||||
if [ \"\$JOB_FILESIZE_Q\" -gt \"\$JOB_COMPLETED_FILESIZE_Q\" ] ; then
|
||||
PBS_JOB4=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_CHRYSALIS.sh\" \`
|
||||
if [[ ! \$PBS_JOB4 ]]; then
|
||||
echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_CHRYSALIS.sh\" FAILED. Aborting...\"
|
||||
exit 255
|
||||
fi
|
||||
let SUBMITTED_C++
|
||||
echo \$PBS_JOB4 >> \"\$JOB_CHRYSALIS.jobnames\"
|
||||
echo \$PBS_JOB4 >> jobnumbers.out ;
|
||||
PBS_JOB5=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED -W depend=afterok:$PBS_JOB4 \"\$JOB_BUTTERFLY.sh\" \`
|
||||
if [[ ! \$PBS_JOB5 ]]; then
|
||||
echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED -W depend=afterok:$PBS_JOB4 \"\$JOB_BUTTERFLY.sh\" FAILED. Aborting...\"
|
||||
exit 255
|
||||
fi
|
||||
let SUBMITTED_B++
|
||||
echo \$PBS_JOB5 >> \"\$JOB_BUTTERFLY.jobnames\"
|
||||
echo \$PBS_JOB5 >> jobnumbers.out ;
|
||||
if [ \$(( \$JOBID % 20 )) -eq 0 ] ; then
|
||||
echo Submitting up to 20 Quantify and/or Butterfly jobs
|
||||
sleep 3 # be nice
|
||||
fi
|
||||
# else all Q have completed; have B completed?
|
||||
else
|
||||
# if at least some B have completed
|
||||
if [ -s \"\$MYJOBB.completed\" ] ; then
|
||||
JOB_COMPLETED_FILESIZE_B=\$(stat -c%s \"\$MYJOBB.completed\" )
|
||||
# if not all, run them with no dependency (Q has completed)
|
||||
if [ \"\$JOB_FILESIZE_B\" -gt \"\$JOB_COMPLETED_FILESIZE_B\" ] ; then
|
||||
PBS_JOB5=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_BUTTERFLY.sh\" \`
|
||||
if [[ ! \$PBS_JOB5 ]]; then
|
||||
echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_BUTTERFLY.sh\" FAILED. Aborting...\"
|
||||
exit 255
|
||||
fi
|
||||
let SUBMITTED_B++
|
||||
echo \$PBS_JOB5 >> \"\$JOB_BUTTERFLY.jobnames\"
|
||||
echo \$PBS_JOB5 >> jobnumbers.out ;
|
||||
if [ \$(( \$JOBID % 20 )) -eq 0 ] ; then
|
||||
echo Submitting up to 20 Quantify and/or Butterfly jobs
|
||||
sleep 3 # be nice
|
||||
fi
|
||||
fi
|
||||
# else no Q have completed; run them without dependency
|
||||
else
|
||||
PBS_JOB5=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_BUTTERFLY.sh\" \`
|
||||
if [[ ! \$PBS_JOB5 ]]; then
|
||||
echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_BUTTERFLY.sh\" FAILED. Aborting...\"
|
||||
exit 255
|
||||
fi
|
||||
let SUBMITTED_B++
|
||||
echo \$PBS_JOB5 >> \"\$JOB_BUTTERFLY.jobnames\"
|
||||
echo \$PBS_JOB5 >> jobnumbers.out ;
|
||||
if [ \$(( \$JOBID % 20 )) -eq 0 ] ; then
|
||||
echo Submitting up to 20 Quantify and/or Butterfly jobs
|
||||
sleep 3 # be nice
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
# neither Q (and thus nor B) have ever ran successfully, submit both with a dependency
|
||||
else
|
||||
PBS_JOB4=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_CHRYSALIS.sh\" \`
|
||||
if [[ ! \$PBS_JOB4 ]]; then
|
||||
echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_CHRYSALIS.sh\" FAILED. Aborting...\"
|
||||
exit 255
|
||||
fi
|
||||
let SUBMITTED_C++
|
||||
echo \$PBS_JOB4 >> \"\$JOB_CHRYSALIS.jobnames\"
|
||||
echo \$PBS_JOB4 >> jobnumbers.out ;
|
||||
PBS_JOB5=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED -W depend=afterok:\$PBS_JOB4 \"\$JOB_BUTTERFLY.sh\" \`
|
||||
if [[ ! \$PBS_JOB5 ]]; then
|
||||
echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED -W depend=afterok:\$PBS_JOB4 \"\$JOB_BUTTERFLY.sh\" FAILED. Aborting...\"
|
||||
exit 255
|
||||
fi
|
||||
let SUBMITTED_B++
|
||||
echo \$PBS_JOB5 >> \"\$JOB_BUTTERFLY.jobnames\"
|
||||
echo \$PBS_JOB5 >> jobnumbers.out ;
|
||||
if [ \$(( \$JOBID % 20 )) -eq 0 ] ; then
|
||||
echo Submitting up to 20 Quantify and/or Butterfly jobs
|
||||
sleep 3 # be nice
|
||||
fi
|
||||
fi
|
||||
done
|
||||
|
||||
echo Submitted \$SUBMITTED_C Chrysalis and \$SUBMITTED_B Butterfly jobs
|
||||
|
||||
if [[ \$SUBMITTED_B == 0 && \$SUBMITTED_C == 0 ]]; then
|
||||
echo \"No Trinity jobs need to be submitted \"
|
||||
if [ -s "$OUTPUTDIR"/Trinity.fasta.complete ]; then
|
||||
echo \"Trinity RNA-Seq assembly is complete! Result file is present as "$OUTPUTDIR"/Trinity.fasta \"
|
||||
else
|
||||
echo \"Proceeding with capturing the output with this command\"
|
||||
echo ' find "$OUTPUTDIR"/chrysalis -name *allProbPaths.fasta -exec cat {} \\; > "$OUTPUTDIR"/Trinity.fasta '
|
||||
find "$OUTPUTDIR"/chrysalis -name *allProbPaths.fasta -exec cat {} \\; > "$OUTPUTDIR"/Trinity.fasta
|
||||
touch "$OUTPUTDIR"/Trinity.fasta.complete
|
||||
echo DO: rm -f "$OUTPUTDIR"/bowtie.nameSorted.sam* "$OUTPUTDIR"/both.fa* "$OUTPUTDIR"/inchworm.kmer_count "$OUTPUTDIR"/iworm_* "$OUTPUTDIR"/target* "$OUTPUTDIR"/jellyfish* "$OUTPUTDIR"/scaffolding* "$OUTPUTDIR"/*.finished "$OUTPUTDIR"/mer_counts_*
|
||||
fi
|
||||
fi
|
||||
"
|
||||
|
||||
|
||||
######
|
||||
##### Write the above script to a file for later execution
|
||||
echo "${JOBSTRING4}" | cat -> "$JOBNAME4.sh"
|
||||
@@ -0,0 +1,42 @@
|
||||
##################################################################################################################################
|
||||
########################## ########################################
|
||||
########################## Trinity PBS job submission with multi part dependencies ########################################
|
||||
########################## ########################################
|
||||
##################################################################################################################################
|
||||
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
|
||||
### Version 1.0
|
||||
### QuantifyGraph p4b Script
|
||||
##################################################################################################################################
|
||||
|
||||
JOBSTRING4b=""$HASHBANG"
|
||||
"$NODESCPUS"
|
||||
if [[ ! \$JOB_INDEX_PADDED ]];then
|
||||
echo \"Error: not a proper submission\"
|
||||
exit 255
|
||||
fi
|
||||
echo \"Processing quantifyGraph_commands index \$JOB_INDEX_PADDED \"
|
||||
cd "$OUTPUTDIR"
|
||||
export OMP_NUM_THREADS=1
|
||||
COREFILENAME=""$OUTPUTDIR"/chrysalis/quantifyGraph_commands.pbs"
|
||||
MYJOBQ=\$COREFILENAME.\$JOB_INDEX_PADDED
|
||||
JOB_FILESIZE=\$(stat -c%s \"$MYJOBQ\")
|
||||
if [ -s \"\$MYJOBQ.completed\" ] ; then
|
||||
JOB_COMPLETED_FILESIZE=\$(stat -c%s \"\$MYJOBQ.completed\")
|
||||
if [ \"\$JOB_FILESIZE\" != \"\$JOB_COMPLETED_FILESIZE\" ] ; then
|
||||
trap \" echo \\\"Please check \$MYJOBQ Chrysalis QuantifyGraph processes had enough walltime.\\\"; exit 255 \" INT TERM
|
||||
"$TRINITYPATH"/trinity-plugins/parafly/bin/ParaFly -CPU "$NCPU_P4" -v -failed_cmds \$MYJOBQ.failed -c \$MYJOBQ
|
||||
trap - INT TERM EXIT
|
||||
fi
|
||||
else
|
||||
trap \" echo \\\"Please check \$MYJOBQ Chrysalis QuantifyGraph processes had enough walltime.\\\"; exit 255 \" INT TERM
|
||||
"$TRINITYPATH"/trinity-plugins/parafly/bin/ParaFly -CPU "$NCPU_P4" -v -failed_cmds \$MYJOBQ.failed -c \$MYJOBQ
|
||||
trap - INT TERM EXIT
|
||||
fi
|
||||
|
||||
sleep 30 # IO friendship for following butterfly job - sometimes butterfly fails to find output if io is overwhelmed
|
||||
exit
|
||||
|
||||
"
|
||||
# Write the above script to a file for later execution
|
||||
echo "${JOBSTRING4b}" | cat -> "$JOBPREFIX"_p4b.sh
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
##################################################################################################################################
|
||||
########################## ########################################
|
||||
########################## Trinity PBS job submission with multi part dependencies ########################################
|
||||
########################## ########################################
|
||||
##################################################################################################################################
|
||||
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
|
||||
### Version 1.0
|
||||
### Butterfly p5b Script
|
||||
##################################################################################################################################
|
||||
|
||||
JOBSTRING5b=""$HASHBANG"
|
||||
"$NODESCPUS"
|
||||
if [[ ! \$JOB_INDEX_PADDED ]];then
|
||||
echo \"Error: not a proper submission\"
|
||||
exit 255
|
||||
fi
|
||||
echo \"Processing butterfly_commands index \$JOB_INDEX_PADDED \"
|
||||
cd "$OUTPUTDIR"
|
||||
export OMP_NUM_THREADS=1
|
||||
COREFILENAME=""$OUTPUTDIR"/chrysalis/butterfly_commands.pbs"
|
||||
MYJOBB=\$COREFILENAME.\$JOB_INDEX_PADDED
|
||||
JOB_FILESIZE=\$(stat -c%s \"\$MYJOBB\")
|
||||
if [ -s \"\$MYJOBB.completed\" ] ; then
|
||||
JOB_COMPLETED_FILESIZE=\$(stat -c%s \"\$MYJOBB.completed\")
|
||||
if [ \"$JOB_FILESIZE\" != \"\$JOB_COMPLETED_FILESIZE\" ] ; then
|
||||
trap \" echo \\\"Please check \$MYJOBB Butterfly processes had enough walltime.\\\"; exit 255 \" INT TERM
|
||||
"$TRINITYPATH"/trinity-plugins/parafly/bin/ParaFly -CPU "$NCPU_P5" -v -failed_cmds \$MYJOBB.failed -c \$MYJOBB
|
||||
trap - INT TERM EXIT
|
||||
fi
|
||||
else
|
||||
trap \" echo \\\"Please check \$MYJOBB Butterfly processes had enough walltime.\\\"; exit 255 \" INT TERM
|
||||
"$TRINITYPATH"/trinity-plugins/parafly/bin/ParaFly -CPU "$NCPU_P5" -v -failed_cmds \$MYJOBB.failed -c \$MYJOBB
|
||||
trap - INT TERM EXIT
|
||||
fi
|
||||
exit
|
||||
|
||||
|
||||
"
|
||||
# Write the above script to a file for later execution
|
||||
echo "${JOBSTRING5b}" | cat -> "$JOBPREFIX"_p5b.sh
|
||||
|
||||
@@ -0,0 +1,291 @@
|
||||
#!/bin/bash
|
||||
set -e # turn on exit on error
|
||||
##################################################################################################################################
|
||||
########################## ########################################
|
||||
########################## Trinity PBS job submission with multi part dependencies ########################################
|
||||
########################## ########################################
|
||||
##################################################################################################################################
|
||||
### Author: Josh Bowden, Alexie Papanicolaou, CSIRO
|
||||
### Version 1.0
|
||||
###
|
||||
###
|
||||
### Script to split the Trinity workflow into multiple stages so as to efficiently request
|
||||
### and use appropriate resources (walltime and number of cores) on a computer cluster / supercomputer.
|
||||
### Currently creates scripts for PBS Torque or PBSpro
|
||||
###
|
||||
### trinity_pbs script install instructions:
|
||||
### 1. Copy all trinity_pbs.* files into a directory (we will call it "TRINITY_PBS_DIR").
|
||||
### 2. Add TRINITY_PBS_DIR to the PATH i.e. export or set PATH=TRINITY_PBS_DIR:$PATH (perhaps export PATH in .bashrc file)
|
||||
### 3. Change the "TRINITYPBSPATH" variable found below to point to the directory also. i.e. TRINITYPBSPATH=TRINITY_PBS_DIR
|
||||
### 4. Set MEMDIRIN to name of a node-local filesystem so a network drive is not needed unecesarily for Scripts 4b and 5b
|
||||
### 5. Set MODTRINITY to any modules that need to be loaded so Trinity.pl can be run.
|
||||
### 6. Set TRINITYPATH to the path to Trinity.pl executable
|
||||
### 7. Set PBSTYPE to --pbspro or --pbs, dependent on the system present.
|
||||
### That should be all that is needed from an admin perspective (besides making scripts accessible and exectable for users)
|
||||
###
|
||||
### Users need make a copy of TRINITY.CONFIG.template and then modify variables in it. See TRINITY.CONFIG.template for further details.
|
||||
###
|
||||
### The current script does the following.
|
||||
### Part 1. Reads data from TRINITY.CONFIG and creates the input directory, data file names, output data directory and Trinity.pl command line
|
||||
### User inputs from TRINITY.CONFIG file :
|
||||
### JOBPREFIX A string of less than 11 characters long. PBS will use this as a jobname prefix.
|
||||
### DATADIRECTORY Where input data exists
|
||||
### OUTPUTDIR Where user wants output data to go - requires a lot of space even for small datatsets
|
||||
### STANDARD_JOB_DETAILS the Trinity.pl command line
|
||||
### ACCOUNT Account details of user (if required by PBS system being used)
|
||||
###
|
||||
### Part 2. Writes scripts to run Trinity.pl in 6 stages:
|
||||
### 3 intial (Inchworm, and 2 x Chrysalis stages: Chrysalis::GraphFromFasta and Chrysalis::ReadsFromTranscripts)
|
||||
### 2 parallel stages (Chrysalis::QuantifyGraph and Butterfly) which are executed in parallel.
|
||||
### 1 collection of results as Trinity.Fasta.
|
||||
###
|
||||
### Information input from command line filename for stage 'x' :
|
||||
### WALLTIME_Px Amount of time stage requires
|
||||
### MEM_Px The amount of memory the stage requires
|
||||
### NCPU_Px The number of CPUs the stage may use
|
||||
### PBSNODETYPE _Px The PBS (for --pbspro only) queue name
|
||||
### NUMPERARRAYITEM_Px The number of massively parallel jobs in each parallel satge.
|
||||
###
|
||||
### Part 3. Runs scripts dependant upon what stage has been detected as completed, using PBS job dependencies
|
||||
###
|
||||
### Command line usage:
|
||||
### To start (or re-start) an analysis:
|
||||
### >trinity_pbs.sh TRINITY.CONFIG.template
|
||||
### To stop previously started PBS jobs on the queue:
|
||||
### >trinity_kill.pl OUTPUTDIR
|
||||
### Where:
|
||||
### TRINITY.CONFIG.template = user specific job details
|
||||
### OUTPUTDIR = is path to output data directory
|
||||
###
|
||||
### Output job script submission files. These are saved in the output directory (OUTPUTDIR) and can be modified/re-run if any job fails.
|
||||
### *_run.sh Runs all the following scripts - with job dependencies and only the jobs that still need to be run.
|
||||
### *_p1.sh Runs Inchworm stage. Does not scale well past a single socket. Only request at most the number of cores on a single CPU.
|
||||
### *_p2.sh Runs Chrysalis::GrapghFromFasta clustering of Inchworm output. Should scale to number of cores on node
|
||||
### *_p3.sh Runs Chrysalis::ReadsToTranscripts. I/O limited. Try to use local filesystem (not implemented)
|
||||
### *_p4a.sh Creates jobs to run Chrysalis::QuantifyGraph and Butterfly parallel tasks
|
||||
### *_p4b.sh QuantifyGraph job. "NUMPERARRAYITEM" tasks from the file /chrysalis/quantifyGraph_commands are run for each job
|
||||
### *_p5b.sh Butterfly job. "NUMPERARRAYITEM" tasks from the file /chrysalis/butterfly_commands are run for each job
|
||||
### to start off at last completed stage. At present leaves all data on temporary area of shared network drive and
|
||||
### copies Trinity.fatsa to home directory (with specific job prefix in filename).
|
||||
### * = $JOBPREFIX. $JOBPREFIX should not be > 10 characters long
|
||||
|
||||
|
||||
##################################################################################################################################
|
||||
####################### SET TRINITY INSTALLATION PATH (where Trinity.pl resides #################################################
|
||||
##################################################################################################################################
|
||||
# We need the path even if loaded using module
|
||||
TRINITYPATH="/home/pap056/software/trinity_2013_08_14/"
|
||||
# If you are loading using module, you can make the next variable blank
|
||||
NEWPATH=
|
||||
NEWPATH="export PATH=$PATH:$TRINITYPATH"
|
||||
|
||||
##################################################################################################################################
|
||||
######## Set TRINITYPBSPATH to the directory where trinity_pbs.sh scripts are installed
|
||||
##################################################################################################################################
|
||||
TRINITYPBSPATH=`dirname "$0"`; # set to location of this script
|
||||
|
||||
##################################################################################################################################
|
||||
######## Set cluster specific name for compute node local filesystem
|
||||
##################################################################################################################################
|
||||
MEMDIRIN="\$TMPDIR" # available on Barrine
|
||||
|
||||
|
||||
##################################################################################################################################
|
||||
######## Set system specific PATHS and load system specific modules (if available)
|
||||
##################################################################################################################################
|
||||
# Example for for Barrine:
|
||||
PBSTYPE="--pbspro"
|
||||
# Here we ensure that Java 1.6 is used and Java 1.7 is removed (Butterfly dependency)
|
||||
MODTRINITY="
|
||||
module load mpt/2.00 perl/5.15.8 bowtie/12.7 jellyfish/1.1.5 samtools/1.18 java/1.6.0_22-sun;
|
||||
module rm java/1.7.0_02
|
||||
"
|
||||
|
||||
## That should be all the admin modifications needed.
|
||||
|
||||
# Append Trinity path data to env variables loaded by every script
|
||||
MODTRINITY="
|
||||
$NEWPATH;
|
||||
export TRINITYPATH="/home/pap056/software/trinity_2013_08_14";
|
||||
$MODTRINITY
|
||||
"
|
||||
##################################################################################################################################
|
||||
##################################################################################################################################
|
||||
##################################################################################################################################
|
||||
|
||||
##################################################################################################################################
|
||||
######### Load files that contains functions
|
||||
##################################################################################################################################
|
||||
# Modify function F_GETNODESTRING in file trinity_pbs.header so that a correct PBS header is returned to suit your PBS cluster
|
||||
if [ -e "$TRINITYPBSPATH"/trinity_pbs.header ] ; then
|
||||
source "$TRINITYPBSPATH"/trinity_pbs.header
|
||||
else
|
||||
echo "$1 requires file \"trinity_pbs.header\" to be present in: "
|
||||
echo "$TRINITYPBSPATH"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
##################################################################################################################################
|
||||
######### Load input config file
|
||||
##################################################################################################################################
|
||||
if [ -e "$1" ] ; then
|
||||
source "$1"
|
||||
else
|
||||
echo "Error: Input file does not exist: "$1" "
|
||||
exit 1
|
||||
fi
|
||||
|
||||
|
||||
##################################################################################################################################
|
||||
## Common variables to PBS and PBSpro
|
||||
## and other needed variables that a user should not need to modify
|
||||
##################################################################################################################################
|
||||
if [ $UEMAIL ]; then PBSUSER="#PBS -M "$UEMAIL"" ; fi
|
||||
HASHBANG="#!/bin/bash"
|
||||
|
||||
##################################################################################################################################
|
||||
## PBS torque and PBSpro have some differences.
|
||||
## Organise these here and also check further on (line 184) and change NODETYPE to match cluster system
|
||||
## MODTRINITY will also be different on different clusters - it sets up the paths to the required executables
|
||||
##################################################################################################################################
|
||||
if [[ "$PBSTYPE" = "--pbspro" ]] ; then
|
||||
JOBARRAY="-J"
|
||||
JOBARRAY_ID="\$PBS_ARRAY_INDEX"
|
||||
AFTEROKARRAY="afterok"
|
||||
elif [[ "$PBSTYPE" = "--pbs" ]] ; then
|
||||
## PBS torque:
|
||||
JOBARRAY="-t"
|
||||
JOBARRAY_ID="\$PBS_ARRAYID"
|
||||
AFTEROKARRAY="afterokarray"
|
||||
else # no paramaters present
|
||||
F_USAGE
|
||||
exit 0
|
||||
fi
|
||||
|
||||
##############################################################################################################################################
|
||||
########## Part 1: Set up file names for input directory and for output data dir and Trinity.pl command line ####################
|
||||
##############################################################################################################################################
|
||||
echo ""
|
||||
## Ensure JOBPREFIX is not greatr than 11 characters as PBS-pro can not handle > 15 characters for total job name length
|
||||
echo "submitting trinity jobs with prefix: "
|
||||
echo " $JOBPREFIX"
|
||||
|
||||
###### Set input data directory - $DATADIR is CSIRO specific
|
||||
echo "Input directory: "
|
||||
echo " $DATADIRECTORY"
|
||||
|
||||
###### Set output data directory (OUTPUTDIR)
|
||||
echo "Output directory: Scripts and output data will be written to:"
|
||||
echo " $OUTPUTDIR"
|
||||
mkdir -p "$OUTPUTDIR"
|
||||
cd "$OUTPUTDIR"
|
||||
|
||||
|
||||
### Modify STANDARD_JOB_DETAILS for analysis specific input to Trinity.pl
|
||||
echo "The following trinity command line will be run:"
|
||||
echo "$STANDARD_JOB_DETAILS"
|
||||
echo ""
|
||||
if [[ "$1" = "--pbs" ]] ; then
|
||||
echo " Use: \"pbs_check.pl -t PBS_JOBID\" "
|
||||
echo " To view stdout and stderr from each separate job while they are running"
|
||||
fi
|
||||
###########################################################################################################################################################
|
||||
### Do some checking that files exist etc. (User should not modify)
|
||||
### This sets the $DS variable
|
||||
|
||||
SET_DS "$FILENAMEINPUT"
|
||||
|
||||
###########################################################################################################################################################
|
||||
################################# ###################################
|
||||
################################# Part 2: Create the shell scripts to be run via the PBS batch system ###################################
|
||||
################################# Users should modify WALLTIME and MEM dependent upon dataset size ###################################
|
||||
################################# and NCPU to appropriate value for compute node cpu resources ###################################
|
||||
################################# Check: "Trinity RNA-seq Assembler Performance Optimisation" (Henschel 2012) ###################################
|
||||
################################# for current best practice. ###################################
|
||||
################################# N.B. On busy clusters it may be best not to try to request ###################################
|
||||
################################# a full nodes resources. i.e. If 8 cores per node are present, ###################################
|
||||
################################# only request half of these. ###################################
|
||||
###########################################################################################################################################################
|
||||
|
||||
###########################################################################################################################################################
|
||||
############################## Script 1: Write script to run Inchworm ############################################
|
||||
############################# MEM should equal JFMEM, which is the amount of memory requested for Jellyfish ############################################
|
||||
|
||||
JOBNAME1="$JOBPREFIX"_p1
|
||||
NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P1" "$NCPU_P1" "$PBSNODETYPE_P1" "$WALLTIME_P1" "$JOBNAME1" "$ACCOUNT" "$PBSUSER" "$MODTRINITY" "$JOBPREFIX")
|
||||
|
||||
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p1"
|
||||
|
||||
#############################################################################################################################################################
|
||||
############################## Script 2: Chrysalis::GraphFromFasta #############################################
|
||||
############################## This script has a dependency on part 1 completion without error. #############################################
|
||||
# It would be good to force an exit(0) before ReadsToTranscripts after checkpoint file /chrysalis/GraphFromIwormFasta.finished is
|
||||
# written (i.e. add --no_run_readstotrans to Trinity and pass through to Chrysalis) as the script 3 can be started directly after.
|
||||
# This may be less of an issue with the new (fast) version of Trinity::GraphFromFasta (since version 2012-06-08).
|
||||
|
||||
|
||||
JOBNAME2="$JOBPREFIX"_p2
|
||||
NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P2" "$NCPU_P2" "$PBSNODETYPE_P2" "$WALLTIME_P2" "$JOBNAME2" "$ACCOUNT" "$PBSUSER" "$MODTRINITY" "$JOBPREFIX")
|
||||
|
||||
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p2"
|
||||
|
||||
###########################################################################################################################################################
|
||||
############################## Script 3: Script to run Chrysalis::ReadsToTranscripts ###################
|
||||
############################## ReadsToTranscripts can be slow due to reads from disk, ###################
|
||||
############################## This script has a dependency on part 2 completion with error. ###################
|
||||
############################## Section script is skipped if enough time was given in Part 2. ###################
|
||||
|
||||
|
||||
JOBNAME3="$JOBPREFIX"_p3
|
||||
NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P3" "$NCPU_P3" "$PBSNODETYPE_P3" "$WALLTIME_P3" "$JOBNAME3" "$ACCOUNT" "$PBSUSER" "$MODTRINITY" "$JOBPREFIX")
|
||||
|
||||
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p3"
|
||||
|
||||
|
||||
##########################################################################################################################################################
|
||||
############################## Script 4a: Write script to call the Chrysalis QuantifyGraph array job ##################
|
||||
############################## N.B. SLOTLIMIT="%x" indicates 'slot limit' i.e. the number of concurrent jobs to execute in an array ##################
|
||||
############################## (Not available in PBSpro ) ##################
|
||||
#### NB Disabling emails for arrays
|
||||
|
||||
#SLOTLIMIT="%64" # available for PBS Torque
|
||||
JOBNAME4="$JOBPREFIX"_p4a
|
||||
NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" 1gb 1 "$PBSNODETYPE_P4" 00:30:00 "$JOBNAME4" "$ACCOUNT" "$PBSUSER" "$MODTRINITY" "$JOBPREFIX" )
|
||||
|
||||
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p4a"
|
||||
|
||||
###########################################################################################################################################################
|
||||
############################## Script Array part 4b: Write script to be run as an array Job . Runs Chrysalis::QuantifyGraph ##################
|
||||
############################## This scipt has a dependency on part 4a being run. If an array component fails it will email user. ##################
|
||||
############################## Files named quantifyGraph_commands_X are written with subset of total commands (X is the array ID from the PBS system) ##
|
||||
|
||||
|
||||
JOBNAME4B="$JOBPREFIX"_p4b
|
||||
NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P4" "$NCPU_P4" "$PBSNODETYPE_P4" "$WALLTIME_P4" "$JOBNAME4B" "$ACCOUNT" " " "$MODTRINITY" "$JOBPREFIX")
|
||||
|
||||
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p4b"
|
||||
|
||||
|
||||
###########################################################################################################################################################
|
||||
############################## Script Array 5b: Write script to run Butterfly Array Job ##################
|
||||
############################## This script has a dependency on part 4a 4b being run. ##################
|
||||
############################## Files named butterfly_commands_X are written with subset of total commands (X is the array ID from the PBS system) ###
|
||||
|
||||
JOBNAME5B="$JOBPREFIX"_p5b
|
||||
NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P5" "$NCPU_P5" "$PBSNODETYPE_P5" "$WALLTIME_P5" "$JOBNAME5B" "$ACCOUNT" " " "$MODTRINITY" "$JOBPREFIX")
|
||||
|
||||
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p5b"
|
||||
|
||||
############################################################################################################################################################
|
||||
############### ###################
|
||||
############### Part 3: Write main control script that executes scripts that were created above. ###################
|
||||
############### ###################
|
||||
############################################################################################################################################################
|
||||
F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.cont"
|
||||
|
||||
############### Run the script written in Part 3
|
||||
############### Checks to see what is current stage of calculation and executes scripts created in above code ###################
|
||||
bash ""$JOBPREFIX"_run.sh"
|
||||
|
||||
exit 0
|
||||
|
||||
Reference in New Issue
Block a user