20251125
This commit is contained in:
@@ -0,0 +1,46 @@
|
||||
#!/usr/bin/env perl
|
||||
|
||||
use strict;
|
||||
use warnings;
|
||||
|
||||
my $usage = "usage: $0 file.audit_summary\n\n";
|
||||
|
||||
my $filename = $ARGV[0] or die $usage;
|
||||
|
||||
main: {
|
||||
|
||||
my %yes_no_counter;
|
||||
my $total_reco = 0;
|
||||
my $total_ref = 0;
|
||||
my $total_FL = 0;
|
||||
|
||||
open (my $fh, $filename) or die $!;
|
||||
while (<$fh>) {
|
||||
if (/ref_fa/) { next; }
|
||||
chomp;
|
||||
my @x = split(/\t/);
|
||||
my $yes_or_no = $x[4];
|
||||
|
||||
$yes_no_counter{$yes_or_no}++;
|
||||
|
||||
my $num_reco = $x[3];
|
||||
$total_reco += $num_reco;
|
||||
|
||||
my $num_ref = $x[1];
|
||||
$total_ref += $num_ref;
|
||||
|
||||
my $num_FL = $x[2];
|
||||
$total_FL += $num_FL;
|
||||
|
||||
}
|
||||
close $fh;
|
||||
|
||||
my $num_yes = $yes_no_counter{YES} || 0;
|
||||
my $num_no = $yes_no_counter{NO} ||0;
|
||||
|
||||
my $num_additional = $total_reco - $total_FL;
|
||||
print join("\t", "#YES", "#NO", "#FL", "#TOT_Trans", "#Ref", "#additional") . "\n";
|
||||
print join("\t", $num_yes, $num_no, $total_FL, $total_reco, $total_ref, $num_additional) . "\n";
|
||||
|
||||
exit(0);
|
||||
}
|
||||
+159
@@ -0,0 +1,159 @@
|
||||
#!/usr/bin/env perl
|
||||
|
||||
use strict;
|
||||
use warnings;
|
||||
use File::Basename;
|
||||
|
||||
use lib ($ENV{EUK_MODULES});
|
||||
use Fasta_reader;
|
||||
|
||||
my $usage = "\n\n\tusage: $0 < list of audit.txt files from stdin \n\n\n";
|
||||
|
||||
my $MIN_SEQ_LEN = 1000;
|
||||
|
||||
|
||||
|
||||
my $total_genes = 0;
|
||||
my $total_genes_reco = 0;
|
||||
|
||||
my $total_refseq_trans = 0;
|
||||
my $total_iso_reco_count = 0;
|
||||
my $total_extra = 0;
|
||||
|
||||
print join("\t", "gene_id", "reco_gene_flag", "num_refseqs", "num_FL_trin", "num_extra_trin") . "\n";
|
||||
|
||||
my $counter = 0;
|
||||
while (<>) {
|
||||
chomp;
|
||||
my $dir = dirname($_);
|
||||
|
||||
$counter++;
|
||||
|
||||
my $gene_id = basename($dir);
|
||||
|
||||
my $trin_fasta_file = "$dir/trinity_out_dir.Trinity.fasta";
|
||||
|
||||
if (! -e $trin_fasta_file) {
|
||||
print STDERR "ERROR: Cannot locate file: $trin_fasta_file\n";
|
||||
next;
|
||||
}
|
||||
|
||||
my $fasta_reader = new Fasta_reader($trin_fasta_file);
|
||||
my %trin_seqs = $fasta_reader->retrieve_all_seqs_hash();
|
||||
|
||||
my %trin_lens = &get_seq_lens(%trin_seqs);
|
||||
|
||||
my $FL_reco_file = "$dir/FL.test.pslx.maps";
|
||||
my %reco_refseq;
|
||||
my %reco_trin_to_refseq = &parse_reco($FL_reco_file, \%reco_refseq);
|
||||
my $num_FL_trin = scalar(keys %reco_trin_to_refseq);
|
||||
|
||||
my $refseqs_fa = "$dir/refseqs.fa";
|
||||
my @refseq_accs = &get_accs($refseqs_fa);
|
||||
|
||||
my $num_refseqs = scalar(@refseq_accs);
|
||||
|
||||
my @failed_reco_refseqs;
|
||||
|
||||
foreach my $acc (@refseq_accs) {
|
||||
unless ($reco_refseq{$acc}) {
|
||||
push (@failed_reco_refseqs, $acc);
|
||||
}
|
||||
}
|
||||
my $num_failed_reco_refseqs = scalar(@failed_reco_refseqs);
|
||||
|
||||
my @extra_trin_accs;
|
||||
foreach my $trin_acc (keys %trin_seqs) {
|
||||
if (! exists $reco_trin_to_refseq{$trin_acc}) {
|
||||
if ($trin_lens{$trin_acc} >= $MIN_SEQ_LEN) {
|
||||
push (@extra_trin_accs, $trin_acc);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
my $num_extra_trin = scalar(@extra_trin_accs);
|
||||
|
||||
my $reco_gene_flag = ($num_failed_reco_refseqs == 0) ? "YES" : "NO";
|
||||
|
||||
print join("\t", $gene_id, $reco_gene_flag, $num_refseqs, $num_FL_trin, $num_extra_trin) . "\n";
|
||||
|
||||
$total_genes++;
|
||||
if ($reco_gene_flag eq "YES") {
|
||||
$total_genes_reco++;
|
||||
}
|
||||
$total_refseq_trans += $num_refseqs;
|
||||
$total_iso_reco_count += $num_FL_trin;
|
||||
$total_extra += $num_extra_trin;
|
||||
|
||||
}
|
||||
|
||||
if ($counter > 1) {
|
||||
print "\n\n";
|
||||
print join("\t", "Total_Genes", "Total_Genes_Reco", "Total_RefTrans", "Total_RefTransReco", "Total_extra_trans") . "\n";
|
||||
print join("\t", $total_genes, $total_genes_reco, $total_refseq_trans, $total_iso_reco_count, $total_extra) . "\n";
|
||||
}
|
||||
|
||||
|
||||
exit(0);
|
||||
|
||||
|
||||
####
|
||||
sub get_accs {
|
||||
my ($fasta_file) = @_;
|
||||
|
||||
my @accs;
|
||||
|
||||
open (my $fh, $fasta_file) or die $!;
|
||||
while (<$fh>) {
|
||||
if (/^>(\S+)/) {
|
||||
push (@accs, $1);
|
||||
}
|
||||
}
|
||||
|
||||
return(@accs);
|
||||
}
|
||||
|
||||
|
||||
####
|
||||
sub parse_reco {
|
||||
my ($reco_file, $reco_refseq_href) = @_;
|
||||
|
||||
my %trin_to_reco_acc;
|
||||
|
||||
open (my $fh, $reco_file) or die $!;
|
||||
while (<$fh>) {
|
||||
chomp;
|
||||
my ($trans_acc, $trinity_contigs) = split(/\t/);
|
||||
|
||||
my @trin_contigs = split(/,/, $trinity_contigs);
|
||||
foreach my $trin (@trin_contigs) {
|
||||
|
||||
$trin_to_reco_acc{$trin}->{$trans_acc} = 1;
|
||||
|
||||
$reco_refseq_href->{$trans_acc} = 1;
|
||||
}
|
||||
}
|
||||
close $fh;
|
||||
|
||||
return(%trin_to_reco_acc);
|
||||
|
||||
}
|
||||
|
||||
|
||||
####
|
||||
sub get_seq_lens {
|
||||
my (%trin_seqs) = @_;
|
||||
|
||||
my %lens;
|
||||
|
||||
foreach my $acc (keys %trin_seqs) {
|
||||
my $seq = $trin_seqs{$acc};
|
||||
my $seqlen = length($seq);
|
||||
|
||||
$lens{$acc} = $seqlen;
|
||||
}
|
||||
|
||||
return(%lens);
|
||||
|
||||
}
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
#!/usr/bin/env perl
|
||||
|
||||
use strict;
|
||||
use warnings;
|
||||
use Cwd;
|
||||
use FindBin;
|
||||
|
||||
my $usage = "usage: $0 info_files.list.txt output_basedir [eval cmds]\n\n";
|
||||
|
||||
my $files_listing_file = $ARGV[0] or die $usage;
|
||||
my $output_basedir = $ARGV[1] or die $usage;
|
||||
shift @ARGV;
|
||||
shift @ARGV;
|
||||
|
||||
|
||||
main: {
|
||||
|
||||
|
||||
my @files = `cat $files_listing_file`;
|
||||
chomp @files;
|
||||
|
||||
my $eval_script = "$FindBin::Bin/run_Trinity_eval.sh";
|
||||
my $basedir = cwd();
|
||||
|
||||
unless ($output_basedir =~ /^\//) {
|
||||
$output_basedir = "$basedir/$output_basedir";
|
||||
}
|
||||
|
||||
foreach my $file (@files) {
|
||||
|
||||
my $line = `cat $file`;
|
||||
chomp $line;
|
||||
my ($refseq_fa_file, $left_fa, $right_fa) = split(/\t/, $line);
|
||||
|
||||
my @pts = split(/\//, $refseq_fa_file);
|
||||
my $gene_name = $pts[-2];
|
||||
|
||||
my $cmd = "$eval_script -R $refseq_fa_file --left $left_fa --right $right_fa -O $output_basedir/$gene_name @ARGV";
|
||||
|
||||
print "$cmd\n";
|
||||
}
|
||||
|
||||
exit(0);
|
||||
}
|
||||
|
||||
|
||||
+308
@@ -0,0 +1,308 @@
|
||||
#!/usr/bin/env perl
|
||||
|
||||
use strict;
|
||||
use warnings;
|
||||
|
||||
use FindBin;
|
||||
use lib ("$ENV{TRINITY_HOME}/PerlLib/");
|
||||
use Fasta_reader;
|
||||
use Cwd;
|
||||
use Data::Dumper;
|
||||
use Carp;
|
||||
use Getopt::Long qw(:config no_ignore_case bundling pass_through);
|
||||
use List::Util qw (shuffle);
|
||||
|
||||
my $help_flag;
|
||||
|
||||
my $ref_trans_fa;
|
||||
my $MIN_REFSEQ_LENGTH = 100;
|
||||
my $OUT_DIR = "Seqs_dir";
|
||||
|
||||
my $MAX_ISOFORMS = -1;
|
||||
my $MIN_ISOFORMS = 2;
|
||||
|
||||
|
||||
my $usage = <<__EOUSAGE__;
|
||||
|
||||
################################################################################
|
||||
#
|
||||
# * Required:
|
||||
#
|
||||
# --ref_trans|R <string> reference transcriptome
|
||||
#
|
||||
# * Common Opts:
|
||||
#
|
||||
# --by_Gene target all isoforms of a gene at once.
|
||||
# (requires multiple isoforms, ignores single-iso genes)
|
||||
#
|
||||
# --out_dir|O <string> output directory name (default: $OUT_DIR)
|
||||
#
|
||||
# * Misc Opts:
|
||||
#
|
||||
# --min_refseq_length <int> min length for a reference transcript
|
||||
# sequence (default: $MIN_REFSEQ_LENGTH)
|
||||
#
|
||||
# if --by_Gene:
|
||||
#
|
||||
# --min_isoforms <int> default: $MIN_ISOFORMS
|
||||
# --max_isoforms <int> max number of isoforms to test (default: $MAX_ISOFORMS)
|
||||
#
|
||||
# --longest_isoform_only restricts to only single longest isoform per gene.
|
||||
#
|
||||
# --restrict_to_genes <string> file containing lists of gene accessions to restrict to.
|
||||
#
|
||||
############################################################################################
|
||||
|
||||
|
||||
|
||||
__EOUSAGE__
|
||||
|
||||
;
|
||||
|
||||
|
||||
|
||||
my $BY_GENE_FLAG = 0;
|
||||
my $LONGEST_ISOFORM_ONLY_FLAG = 0;
|
||||
|
||||
my $restrict_to_genes_file = "";
|
||||
|
||||
&GetOptions ( 'h' => \$help_flag,
|
||||
|
||||
# required
|
||||
'ref_trans|R=s' => \$ref_trans_fa,
|
||||
|
||||
# optional
|
||||
'out_dir|O=s' => \$OUT_DIR,
|
||||
'min_refseq_length=i' => \$MIN_REFSEQ_LENGTH,
|
||||
|
||||
'by_Gene' => \$BY_GENE_FLAG,
|
||||
|
||||
'max_isoforms=i' => \$MAX_ISOFORMS,
|
||||
'min_isoforms=i' => \$MIN_ISOFORMS,
|
||||
|
||||
'longest_isoform_only' => \$LONGEST_ISOFORM_ONLY_FLAG,
|
||||
|
||||
'restrict_to_genes=s' => \$restrict_to_genes_file,
|
||||
);
|
||||
|
||||
|
||||
|
||||
if ($help_flag) {
|
||||
die $usage;
|
||||
}
|
||||
|
||||
unless ($ref_trans_fa) {
|
||||
die $usage;
|
||||
}
|
||||
|
||||
|
||||
main: {
|
||||
|
||||
my $BASEDIR = cwd();
|
||||
|
||||
|
||||
if ($ref_trans_fa =~ /\.gz$/) {
|
||||
my $unzipped = $ref_trans_fa;
|
||||
$unzipped =~ s/\.gz$//g;
|
||||
if (! -s $unzipped) {
|
||||
&process_cmd("gunzip -c $ref_trans_fa > $unzipped");
|
||||
}
|
||||
|
||||
$ref_trans_fa = $unzipped;
|
||||
}
|
||||
|
||||
my $fasta_reader = new Fasta_reader($ref_trans_fa);
|
||||
my %fasta_seqs = $fasta_reader->retrieve_all_seqs_hash();
|
||||
|
||||
my %reorganized_fasta_seqs = &reorganize_fasta_seqs(\%fasta_seqs, $BY_GENE_FLAG);
|
||||
|
||||
|
||||
my %restricted_genes;
|
||||
if ($restrict_to_genes_file) {
|
||||
my @gene_ids = `cat $restrict_to_genes_file`;
|
||||
chomp @gene_ids;
|
||||
%restricted_genes = map { + $_ => 1 } @gene_ids;
|
||||
}
|
||||
|
||||
my $total_counter = 0;
|
||||
|
||||
my @accs = keys %reorganized_fasta_seqs;
|
||||
|
||||
my %seen;
|
||||
|
||||
foreach my $acc (@accs) {
|
||||
|
||||
if (%restricted_genes && ! exists $restricted_genes{$acc}) {
|
||||
# skipping, not in the restricted list.
|
||||
next;
|
||||
}
|
||||
$seen{$acc} = 1;
|
||||
|
||||
chdir $BASEDIR or die "Error, cannot cd to $BASEDIR";
|
||||
|
||||
|
||||
my $seq_entries_aref = $reorganized_fasta_seqs{$acc};
|
||||
|
||||
my @min_length_targets;
|
||||
foreach my $entry (@$seq_entries_aref) {
|
||||
|
||||
my ($trans_acc, $seq) = ($entry->{acc},
|
||||
$entry->{seq});
|
||||
|
||||
if (length($seq) >= $MIN_REFSEQ_LENGTH && $seq !~ /[^GATC]/i) {
|
||||
push (@min_length_targets, $entry);
|
||||
}
|
||||
}
|
||||
|
||||
unless (@min_length_targets) {
|
||||
print STDERR "No min length targets to pursue for $acc .... skipping.\n";
|
||||
next;
|
||||
}
|
||||
|
||||
unless (-d $OUT_DIR) {
|
||||
mkdir($OUT_DIR) or die $!;
|
||||
}
|
||||
|
||||
@min_length_targets = reverse sort {length($a->{seq}) <=> length($b->{seq}) } @min_length_targets;
|
||||
|
||||
my $num_total_targets = scalar(@min_length_targets);
|
||||
|
||||
if ($BY_GENE_FLAG) {
|
||||
|
||||
if ($LONGEST_ISOFORM_ONLY_FLAG) {
|
||||
@min_length_targets = shift @min_length_targets;
|
||||
}
|
||||
else {
|
||||
|
||||
if ($num_total_targets < $MIN_ISOFORMS) {
|
||||
next;
|
||||
}
|
||||
|
||||
if ($num_total_targets > $MAX_ISOFORMS) {
|
||||
|
||||
@min_length_targets = @min_length_targets[0..($num_total_targets-1)];
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
&prep_seqs($acc, \@min_length_targets);
|
||||
|
||||
$total_counter++;
|
||||
if ($total_counter % 100 == 0) {
|
||||
print STDERR "\n[$total_counter]\n";
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
if (%restricted_genes) {
|
||||
# ensure we got them all
|
||||
for my $seen_acc (keys %seen) {
|
||||
if (exists $restricted_genes{$seen_acc}) {
|
||||
delete $restricted_genes{$seen_acc};
|
||||
}
|
||||
}
|
||||
|
||||
if (%restricted_genes) {
|
||||
die "Error, missing entries for restricted gene list entries: " . Dumper(\%restricted_genes);
|
||||
}
|
||||
else {
|
||||
print STDERR "-all restricted gene entries identified and reported.\n";
|
||||
}
|
||||
}
|
||||
|
||||
print STDERR "\nDone.\n\n";
|
||||
|
||||
exit(0);
|
||||
}
|
||||
|
||||
|
||||
|
||||
####
|
||||
sub prep_seqs {
|
||||
my ($acc, $entries_aref) = @_;
|
||||
|
||||
print STDERR "\r-processing $acc ";
|
||||
|
||||
my $num_entries = scalar(@$entries_aref);
|
||||
|
||||
my $basedir = cwd();
|
||||
|
||||
my $dir_tok = $acc;
|
||||
$dir_tok =~ s/\W/_/g;
|
||||
|
||||
my $workdir = "$OUT_DIR/$dir_tok";
|
||||
|
||||
unless (-d $workdir) {
|
||||
mkdir $workdir or die "Error, cannot mkdir $workdir";
|
||||
}
|
||||
|
||||
my $refseqs_fa = "$workdir/refseqs.fa";
|
||||
|
||||
# write ref fasta seqs.
|
||||
if (! -s "$refseqs_fa") {
|
||||
open (my $ofh, ">$refseqs_fa") or die $!;
|
||||
foreach my $entry (@$entries_aref) {
|
||||
my ($entry_acc, $seq) = ($entry->{acc},
|
||||
$entry->{seq});
|
||||
|
||||
print $ofh ">$entry_acc\n$seq\n";
|
||||
}
|
||||
close $ofh;
|
||||
}
|
||||
|
||||
return;
|
||||
|
||||
}
|
||||
|
||||
|
||||
####
|
||||
sub process_cmd {
|
||||
my ($cmd) = @_;
|
||||
|
||||
print STDERR "CMD: $cmd\n";
|
||||
my $ret = system($cmd);
|
||||
if ($ret) {
|
||||
die "Error, cmd: $cmd died with ret $ret";
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
####
|
||||
sub reorganize_fasta_seqs {
|
||||
my ($fasta_seqs_href, $by_gene_flag) = @_;
|
||||
|
||||
my %reorg_fasta;
|
||||
|
||||
foreach my $acc (sort keys %$fasta_seqs_href) {
|
||||
|
||||
my $seq = uc $fasta_seqs_href->{$acc};
|
||||
|
||||
my $key = $acc;
|
||||
if ($by_gene_flag) {
|
||||
if ($acc =~ /^([^;]+);([^;]+)$/) {
|
||||
my $trans = $1;
|
||||
my $gene = $2;
|
||||
$key = $gene;
|
||||
}
|
||||
elsif ($acc =~ /^([^\|]+)\|([^\|]+)$/) {
|
||||
my $gene = $1;
|
||||
my $trans = $2;
|
||||
$key = $gene;
|
||||
}
|
||||
else {
|
||||
confess "Error, no gene ID extracted from $acc ";
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
push (@{$reorg_fasta{$key}}, { acc => $acc,
|
||||
seq => $seq}
|
||||
);
|
||||
}
|
||||
|
||||
return(%reorg_fasta);
|
||||
|
||||
}
|
||||
|
||||
@@ -0,0 +1,461 @@
|
||||
#!/usr/bin/env perl
|
||||
|
||||
use strict;
|
||||
use warnings;
|
||||
|
||||
use FindBin;
|
||||
use lib ("$ENV{TRINITY_HOME}/PerlLib/");
|
||||
use Fasta_reader;
|
||||
use Cwd;
|
||||
use Data::Dumper;
|
||||
use Carp;
|
||||
use Getopt::Long qw(:config no_ignore_case bundling pass_through);
|
||||
use List::Util qw (shuffle);
|
||||
|
||||
my $VERBOSITY_LEVEL = 10;
|
||||
|
||||
|
||||
my $help_flag;
|
||||
my $ref_trans_fa;
|
||||
my $BFLY_JAR = "$ENV{TRINITY_HOME}/Butterfly/Butterfly.jar";
|
||||
my $INCLUDE_REF_TRANS = 0;
|
||||
my $OUT_DIR = "testing_dir";
|
||||
my $MIN_CONTIG_LENGTH = 200;
|
||||
my $min_per_id = 90;
|
||||
|
||||
|
||||
my $usage = <<__EOUSAGE__;
|
||||
|
||||
################################################################################
|
||||
#
|
||||
# * Required:
|
||||
#
|
||||
# --ref_trans|R <string> reference transcriptome
|
||||
#
|
||||
# --left <string> left reads fa file
|
||||
#
|
||||
# --right <string> right reads fa file
|
||||
#
|
||||
# * Common Opts:
|
||||
#
|
||||
# --bfly_jar|B <string> Butterfly jar file
|
||||
#
|
||||
# --out_dir|O <string> output directory name (default: $OUT_DIR)
|
||||
#
|
||||
# * include FL seq opts:
|
||||
#
|
||||
# --incl_ref_trans include the ref transcript as a long
|
||||
# read (default: off)
|
||||
#
|
||||
# * Misc Opts:
|
||||
#
|
||||
# --incl_ref_dot include dot files for the reference sequences
|
||||
#
|
||||
# -V <int> verbosity level (default: 12)
|
||||
#
|
||||
# --acc <string> restrict to a specific accession (gene or transcript)
|
||||
#
|
||||
# --paired_as_single treat paired reads as single reads
|
||||
#
|
||||
# --min_contig_length <int> minimum contig length for Trinity assembly. default: $MIN_CONTIG_LENGTH
|
||||
#
|
||||
# --strict weld all, no pruning or path merging.
|
||||
#
|
||||
# --no_cleanup no cleaning up of trinity output.
|
||||
#
|
||||
# --strand_specific sets to strand-specific mode (RF)
|
||||
#
|
||||
# --bfly_opts <string> butterfly additional opts
|
||||
#
|
||||
############################################################################################
|
||||
|
||||
|
||||
__EOUSAGE__
|
||||
|
||||
;
|
||||
|
||||
|
||||
my $NO_CLEANUP = 0;
|
||||
|
||||
my $INCLUDE_REF_DOT_FILES = 0;
|
||||
|
||||
my $PAIRED_AS_SINGLE = "";
|
||||
|
||||
my $SHUFFLE = 0;
|
||||
|
||||
my $MAX_ISOFORMS = -1;
|
||||
|
||||
my $STRICT = 0;
|
||||
|
||||
|
||||
my $strand_specific_flag = 0;
|
||||
|
||||
my $left_file = "";
|
||||
my $right_file = "";
|
||||
my $BFLY_OPTS;
|
||||
|
||||
&GetOptions ( 'h' => \$help_flag,
|
||||
|
||||
# required
|
||||
'ref_trans|R=s' => \$ref_trans_fa,
|
||||
|
||||
'left=s' => \$left_file,
|
||||
'right=s' => \$right_file,
|
||||
|
||||
# optional
|
||||
'out_dir|O=s' => \$OUT_DIR,
|
||||
'bfly_jar|B=s' => \$BFLY_JAR,
|
||||
|
||||
'incl_ref_trans' => \$INCLUDE_REF_TRANS,
|
||||
|
||||
'strict' => \$STRICT,
|
||||
|
||||
'V=i' => \$VERBOSITY_LEVEL,
|
||||
|
||||
'incl_ref_dot' => \$INCLUDE_REF_DOT_FILES,
|
||||
|
||||
'paired_as_single' => \$PAIRED_AS_SINGLE,
|
||||
|
||||
'min_contig_length=i' => \$MIN_CONTIG_LENGTH,
|
||||
|
||||
'no_cleanup' => \$NO_CLEANUP,
|
||||
|
||||
'strand_specific' => \$strand_specific_flag,
|
||||
|
||||
'min_per_id=i' => \$min_per_id,
|
||||
|
||||
'bfly_opts=s' => \$BFLY_OPTS,
|
||||
);
|
||||
|
||||
|
||||
if ($help_flag) {
|
||||
die $usage;
|
||||
}
|
||||
|
||||
|
||||
|
||||
unless ($ref_trans_fa && $BFLY_JAR && $left_file && $right_file) {
|
||||
die $usage;
|
||||
}
|
||||
|
||||
|
||||
$NO_CLEANUP = 1; ## NEEDED NOW for iworm and bfy pruning assessment
|
||||
|
||||
unless ($ENV{TRINITY_HOME}) {
|
||||
$ENV{TRINITY_HOME} = "$FindBin::Bin/../../trinityrnaseq/";
|
||||
}
|
||||
|
||||
my $reconstructions_log_file = "$OUT_DIR.reconstruction_summary.txt";
|
||||
|
||||
|
||||
if ($PAIRED_AS_SINGLE) {
|
||||
$PAIRED_AS_SINGLE = "--TREAT_PAIRS_AS_SINGLE";
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
main: {
|
||||
|
||||
my $BASEDIR = cwd();
|
||||
|
||||
unless ($ref_trans_fa =~ /^\//) {
|
||||
$ref_trans_fa = "$BASEDIR/$ref_trans_fa";
|
||||
}
|
||||
|
||||
|
||||
unless (-d $OUT_DIR) {
|
||||
&process_cmd("mkdir -p $OUT_DIR");
|
||||
}
|
||||
chdir $OUT_DIR or die "Error, cannot cd to $OUT_DIR";
|
||||
|
||||
|
||||
|
||||
my ($num_reco_FL, $num_ref_entries, $num_trans_reco,
|
||||
$has_all_iworm_kmers, $has_all_precious_edges,
|
||||
$num_LR_threaded) = &execute_seq_pipe($ref_trans_fa, $left_file, $right_file);
|
||||
# num_FL: number of transcripts reconstructed as full-length
|
||||
# num_entries: number of reference isoforms
|
||||
# num_transcripts: total number of Trinity transcripts reconstructed.
|
||||
|
||||
open (my $ofh, ">audit.txt") or die $!;
|
||||
my $captured_all = ($num_reco_FL == $num_ref_entries) ? "YES" : "NO";
|
||||
|
||||
|
||||
my $header = join("\t", "ref_fa", "num_ref", "num_FL", "num_reco", "captured_all", "iworm_ok", "bfly_edges_ok",
|
||||
"num_LR_threaded", "all_LR_threaded_ok");
|
||||
|
||||
my $all_LR_threaded_ok = ($num_LR_threaded == $num_ref_entries) ? "YES" : "NO";
|
||||
|
||||
my $summary = join("\t", $ref_trans_fa,
|
||||
$num_ref_entries,
|
||||
$num_reco_FL,
|
||||
$num_trans_reco,
|
||||
$captured_all,
|
||||
$has_all_iworm_kmers,
|
||||
$has_all_precious_edges,
|
||||
$num_LR_threaded,
|
||||
$all_LR_threaded_ok);
|
||||
|
||||
$summary = "$header\n$summary\n";
|
||||
|
||||
print $ofh $summary;
|
||||
print $summary;
|
||||
close $ofh;
|
||||
|
||||
&process_cmd("echo " . cwd() . "/audit.txt | $FindBin::Bin/audit_summary_stats.reexamine.pl | tee audit2.txt");
|
||||
|
||||
|
||||
exit(0);
|
||||
}
|
||||
|
||||
|
||||
|
||||
####
|
||||
sub execute_seq_pipe {
|
||||
my ($ref_trans_fa, $left_file, $right_file) = @_;
|
||||
|
||||
|
||||
my $cmd = "ln -sf $ref_trans_fa $left_file $right_file .";
|
||||
&process_cmd($cmd);
|
||||
|
||||
if (-d "trinity_out_dir") {
|
||||
`rm -rf ./trinity_out_dir`;
|
||||
}
|
||||
|
||||
my $num_entries = `grep '>' $ref_trans_fa | wc -l `;
|
||||
$num_entries =~ /(\d+)/ or die "Error, cannot parse number of entries from $ref_trans_fa";
|
||||
$num_entries = $1;
|
||||
|
||||
# run Trinity
|
||||
|
||||
my $bfly_jar_txt = "";
|
||||
if ($BFLY_JAR) {
|
||||
$bfly_jar_txt = " --bfly_jar $BFLY_JAR ";
|
||||
}
|
||||
|
||||
$cmd = "set -o pipefail; $ENV{TRINITY_HOME}/Trinity --seqType fa --max_memory 4G --bflyHeapSpaceMax 10G --max_reads_per_graph 10000000 --group_pairs_distance 10000 --verbose_level 2 --CPU 1 ";
|
||||
if ($NO_CLEANUP) {
|
||||
$cmd .= " --no_cleanup ";
|
||||
}
|
||||
else {
|
||||
$cmd .= " --full_cleanup ";
|
||||
}
|
||||
|
||||
if ($INCLUDE_REF_TRANS) {
|
||||
$cmd .= " --long_reads $ref_trans_fa ";
|
||||
}
|
||||
|
||||
$cmd .= " --left $left_file --right $right_file ";
|
||||
|
||||
if ($strand_specific_flag) {
|
||||
$cmd .= " --SS_lib_type RF ";
|
||||
}
|
||||
|
||||
|
||||
$cmd .= " --CPU 2 $bfly_jar_txt --inchworm_cpu 1 --min_contig_length $MIN_CONTIG_LENGTH --trinity_complete @ARGV";
|
||||
|
||||
|
||||
|
||||
my $bfly_opts = " --bfly_opts \"--generate_intermediate_dot_files -R 1 --generate_intermediate_dot_files $PAIRED_AS_SINGLE --stderr -V $VERBOSITY_LEVEL $BFLY_OPTS\" ";
|
||||
|
||||
if ($STRICT) {
|
||||
$cmd .= " --no_bowtie --chrysalis_debug_weld_all "
|
||||
. " --iworm_opts \"--no_prune_error_kmers --min_assembly_coverage 1 --min_seed_entropy 0 --min_seed_coverage 1 \" ";
|
||||
|
||||
$bfly_opts = " --bfly_opts \"--dont-collapse-snps --no_pruning --no_path_merging --no_remove_lower_ranked_paths --NO_EM_REDUCE --MAX_READ_SEQ_DIVERGENCE=0 --NO_DP_READ_TO_VERTEX_ALIGN --generate_intermediate_dot_files -R 1 -F 100000 --generate_intermediate_dot_files $PAIRED_AS_SINGLE --stderr -V $VERBOSITY_LEVEL $BFLY_OPTS\" ";
|
||||
|
||||
}
|
||||
else {
|
||||
#$cmd .= " --no_bowtie --chrysalis_debug_weld_all ";
|
||||
#$cmd .= " --iworm_opts \" --min_seed_entropy 1 \" --min_glue 1 ";
|
||||
}
|
||||
|
||||
$cmd .= " $bfly_opts 2>&1 | tee trin.log";
|
||||
|
||||
|
||||
{
|
||||
open (my $ofh, ">runTrinity.cmd") or die $!;
|
||||
print $ofh $cmd;
|
||||
close $ofh;
|
||||
}
|
||||
|
||||
&process_cmd($cmd);
|
||||
|
||||
if ($NO_CLEANUP) {
|
||||
rename("trinity_out_dir/Trinity.fasta", "trinity_out_dir.Trinity.fasta");
|
||||
}
|
||||
|
||||
|
||||
## check inchworm kmer content of reference sequences
|
||||
|
||||
&process_cmd("$ENV{TRINITY_HOME}/util/misc/print_kmers.pl $ref_trans_fa 24 > ref_kmers");
|
||||
|
||||
|
||||
my $iworm_file = (-s "trinity_out_dir/inchworm.K25.L25.fa") ? "trinity_out_dir/inchworm.K25.L25.fa" : "trinity_out_dir/inchworm.K25.L25.DS.fa";
|
||||
my $has_all_iworm_kmers = &check_inchworm_kmer_content($ref_trans_fa, $iworm_file);
|
||||
|
||||
|
||||
## examine pruning of precious edges
|
||||
my ($has_all_precious_edges) = &check_pruning("trin.log", "ref_kmers");
|
||||
|
||||
## see if all LR are threaded through the graph
|
||||
my $num_LR_threaded = &count_num_LR_threaded("trin.log");
|
||||
|
||||
if ($INCLUDE_REF_DOT_FILES) {
|
||||
# generate sequence graphs just refseqs
|
||||
$cmd = "$ENV{TRINITY_HOME}/util/misc/Monarch --misc_seqs $ref_trans_fa --graph refseqs.dot";
|
||||
if (! -s "refseqs.dot") {
|
||||
&process_cmd($cmd);
|
||||
}
|
||||
|
||||
# generate sequence graphs just refseqs
|
||||
$cmd = "$ENV{TRINITY_HOME}/util/misc/Monarch --misc_seqs $ref_trans_fa,trinity_out_dir/inchworm.K25.L25.fa --graph refseqs_w_iworm.dot";
|
||||
if (! -s "refseqs_w_iworm.dot") {
|
||||
&process_cmd($cmd);
|
||||
}
|
||||
|
||||
# generate sequence graphs combining all
|
||||
$cmd = "$ENV{TRINITY_HOME}/util/misc/Monarch --misc_seqs $ref_trans_fa,trinity_out_dir/inchworm.K25.L25.fa,trinity_out_dir.Trinity.fasta --graph all_compare.dot";
|
||||
if (! -s "all_compare.dot") {
|
||||
&process_cmd($cmd);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
# compare refseqs to the trinity assemblies
|
||||
$cmd = "$ENV{TRINITY_HOME}/util/misc/illustrate_ref_comparison.pl $ref_trans_fa trinity_out_dir.Trinity.fasta $min_per_id | tee ref_compare.ascii_illus";
|
||||
&process_cmd($cmd);
|
||||
|
||||
|
||||
## get number of transcripts reconstructed:
|
||||
$cmd = "grep '>' trinity_out_dir.Trinity.fasta | wc -l";
|
||||
my $result = `$cmd`;
|
||||
$result =~ s/^\s+//g;
|
||||
my ($num_transcripts, @rest) = split(/\s+/, $result);
|
||||
|
||||
|
||||
|
||||
# reconstruction test
|
||||
$cmd = "$ENV{TRINITY_HOME}/Analysis/FL_reconstruction_analysis/FL_trans_analysis_pipeline.pl --target $ref_trans_fa --query trinity_out_dir.Trinity.fasta --no_reuse --out_prefix FL.test --allow_non_unique_mappings --min_per_length 90 --min_per_id $min_per_id | tee FL_analysis.txt";
|
||||
|
||||
&process_cmd("echo $cmd > FL.cmd");
|
||||
my @results = `$cmd`;
|
||||
print @results;
|
||||
|
||||
chomp @results;
|
||||
shift @results;
|
||||
shift @results;
|
||||
$result = shift @results;
|
||||
$result =~ s/^\s+//;
|
||||
|
||||
my @pts = split(/\s+/, $result);
|
||||
my $num_FL = $pts[2] || 0;
|
||||
|
||||
my $got_all_flag = 0;
|
||||
|
||||
if ($num_FL == $num_entries) {
|
||||
print STDERR "-got all FL ($num_FL reconstructed / $num_entries total reconstructed).\n";
|
||||
|
||||
$got_all_flag = 1;
|
||||
#print STDERR Dumper(\@pts);
|
||||
}
|
||||
else {
|
||||
print STDERR "** missed at least one reconstructed isoform ($num_FL reconstructed / $num_entries total reconstructed).\n";
|
||||
}
|
||||
|
||||
|
||||
|
||||
# cleanup really needed after all.
|
||||
unless ($NO_CLEANUP) {
|
||||
system("rm -rf ./trinity_out_dir");
|
||||
}
|
||||
|
||||
return ($num_FL, $num_entries, $num_transcripts, $has_all_iworm_kmers, $has_all_precious_edges, $num_LR_threaded);
|
||||
|
||||
|
||||
}
|
||||
|
||||
|
||||
####
|
||||
sub process_cmd {
|
||||
my ($cmd) = @_;
|
||||
|
||||
print STDERR "CMD: $cmd\n";
|
||||
my $ret = system($cmd);
|
||||
if ($ret) {
|
||||
die "Error, cmd: $cmd died with ret $ret";
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
|
||||
####
|
||||
sub check_inchworm_kmer_content {
|
||||
my ($ref_trans_fa, $iworm_file) = @_;
|
||||
|
||||
my $cmd = "$ENV{TRINITY_HOME}/Inchworm/bin/inchworm --threadFasta $ref_trans_fa --reads $iworm_file > $iworm_file.ref_kmer_check";
|
||||
&process_cmd($cmd);
|
||||
|
||||
my $has_all_kmers = "YES";
|
||||
|
||||
open (my $fh, "$iworm_file.ref_kmer_check") or die $!;
|
||||
while (<$fh>) {
|
||||
chomp;
|
||||
my @x = split(/\t/);
|
||||
my $kmer_info = $x[1];
|
||||
if ($kmer_info && $kmer_info =~ /:/) {
|
||||
my ($kmer, $count) = split(/:/, $kmer_info);
|
||||
if ($count == 0) {
|
||||
print "Inchworm missing kmer: $kmer\n";
|
||||
$has_all_kmers = "NO";
|
||||
}
|
||||
}
|
||||
}
|
||||
close $fh;
|
||||
|
||||
return($has_all_kmers);
|
||||
}
|
||||
|
||||
####
|
||||
sub check_pruning {
|
||||
my ($log_file, $ref_kmers_file) = @_;
|
||||
|
||||
my $pruned_ref_edges = `$FindBin::Bin/util/find_pruned_edges_shouldve_kept.pl $ref_kmers_file $log_file`;
|
||||
|
||||
if ($pruned_ref_edges =~ /\w/) {
|
||||
print "Pruned precious edges: $pruned_ref_edges\n";
|
||||
return("NO");
|
||||
}
|
||||
else {
|
||||
print "no pruning of precious edges\n";
|
||||
return("YES");
|
||||
}
|
||||
}
|
||||
|
||||
####
|
||||
sub count_num_LR_threaded {
|
||||
my ($logfile) = @_;
|
||||
|
||||
# FINAL BEST PATH for LR$|ENST00000479454.1;ASZ1_mutated is [1, 228, 373, 2906, 598, 2885, 771] with total mm: 55
|
||||
# No read mapping found for: LR$|ENST00000465832.1;ASZ1_mutated
|
||||
|
||||
my $num_LR_threaded = 0;
|
||||
|
||||
open(my $fh, $logfile) or die "Error, cannot open file $logfile";
|
||||
while(<$fh>) {
|
||||
chomp;
|
||||
if (/FINAL BEST PATH for LR\$/) {
|
||||
print "$_\n";
|
||||
$num_LR_threaded++;
|
||||
}
|
||||
elsif (/No read mapping found for: LR\$/) {
|
||||
print "$_\n";
|
||||
}
|
||||
}
|
||||
close $fh;
|
||||
|
||||
return($num_LR_threaded);
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
#!/bin/bash
|
||||
|
||||
#source /broad/software/scripts/useuse
|
||||
|
||||
#reuse Perl-5.8
|
||||
#reuse .samtools-0.1.19
|
||||
#reuse GCC-4.9
|
||||
|
||||
|
||||
|
||||
CMD="`dirname $0`/run_Trinity_eval.pl $*"
|
||||
|
||||
eval $CMD
|
||||
|
||||
exit $?
|
||||
|
||||
@@ -0,0 +1,148 @@
|
||||
#!/usr/bin/env perl
|
||||
|
||||
use strict;
|
||||
use warnings;
|
||||
|
||||
use FindBin;
|
||||
use lib ("$ENV{TRINITY_HOME}/PerlLib/");
|
||||
use Fasta_reader;
|
||||
use Cwd;
|
||||
use Data::Dumper;
|
||||
use Carp;
|
||||
use Getopt::Long qw(:config no_ignore_case bundling pass_through);
|
||||
use List::Util qw (shuffle);
|
||||
|
||||
my $help_flag;
|
||||
|
||||
|
||||
my $usage = <<__EOUSAGE__;
|
||||
|
||||
################################################################################
|
||||
|
||||
$0
|
||||
|
||||
################################################################################
|
||||
#
|
||||
# * Required:
|
||||
#
|
||||
# --ref_trans|R <string> reference transcriptome
|
||||
#
|
||||
# --out_dir|O <string> output directory name
|
||||
#
|
||||
# --read_length <int> default: 76
|
||||
#
|
||||
# --frag_length <int> default: 300
|
||||
#
|
||||
# --depth_of_cov <int> default: 100
|
||||
#
|
||||
#
|
||||
####
|
||||
#
|
||||
# following wgsim options are pass-through:
|
||||
#
|
||||
# Options:
|
||||
# -e FLOAT base error rate [0.020]
|
||||
# -s INT standard deviation [50]
|
||||
# -r FLOAT rate of mutations [0.0010]
|
||||
# -R FLOAT fraction of indels [0.15]
|
||||
# -X FLOAT probability an indel is extended [0.30]
|
||||
# -S INT seed for random generator [-1]
|
||||
# -A FLOAT disgard if the fraction of ambiguous bases higher than FLOAT [0.05]
|
||||
# -h haplotype mode
|
||||
# -Z INT strand specific mode: 1=FR, 2=RF
|
||||
# -D debug mode... highly verbose
|
||||
#
|
||||
#
|
||||
############################################################################################
|
||||
|
||||
|
||||
__EOUSAGE__
|
||||
|
||||
;
|
||||
|
||||
|
||||
my $OUT_DIR;
|
||||
my $ref_trans_fa;
|
||||
my $read_length = 76;
|
||||
my $frag_length = 300;
|
||||
my $depth_of_cov = 100;
|
||||
|
||||
|
||||
&GetOptions ( 'help' => \$help_flag,
|
||||
|
||||
# required
|
||||
'ref_trans|R=s' => \$ref_trans_fa,
|
||||
|
||||
# optional
|
||||
'out_dir|O=s' => \$OUT_DIR,
|
||||
|
||||
'read_length=i' => \$read_length,
|
||||
'frag_length=i' => \$frag_length,
|
||||
'depth_of_cov=i' => \$depth_of_cov,
|
||||
|
||||
);
|
||||
|
||||
|
||||
if ($help_flag) {
|
||||
die $usage;
|
||||
}
|
||||
|
||||
|
||||
unless ($ref_trans_fa && $OUT_DIR) {
|
||||
die $usage;
|
||||
}
|
||||
|
||||
|
||||
|
||||
unless ($ENV{TRINITY_HOME}) {
|
||||
$ENV{TRINITY_HOME} = "$FindBin::Bin/../../trinityrnaseq/";
|
||||
}
|
||||
|
||||
|
||||
main: {
|
||||
|
||||
my $BASEDIR = cwd();
|
||||
|
||||
unless ($ref_trans_fa =~ /^\//) {
|
||||
$ref_trans_fa = "$BASEDIR/$ref_trans_fa";
|
||||
}
|
||||
|
||||
|
||||
unless (-d $OUT_DIR) {
|
||||
&process_cmd("mkdir -p $OUT_DIR");
|
||||
}
|
||||
chdir $OUT_DIR or die "Error, cannot cd to $OUT_DIR";
|
||||
|
||||
|
||||
my $cmd = "";
|
||||
|
||||
|
||||
# simulate reads:
|
||||
$cmd = "$ENV{TRINITY_HOME}/util/misc/simulate_illuminaPE_from_transcripts.wgsim.pl --transcripts $ref_trans_fa "
|
||||
. " --read_length $read_length "
|
||||
. " --frag_length $frag_length "
|
||||
. " --depth_of_cov 200 "
|
||||
. " @ARGV "; # wgsim opts pass-through
|
||||
;
|
||||
|
||||
## todo: add mutation rate info
|
||||
|
||||
&process_cmd($cmd);
|
||||
|
||||
}
|
||||
|
||||
|
||||
|
||||
####
|
||||
sub process_cmd {
|
||||
my ($cmd) = @_;
|
||||
|
||||
print STDERR "CMD: $cmd\n";
|
||||
my $ret = system($cmd);
|
||||
if ($ret) {
|
||||
die "Error, cmd: $cmd died with ret $ret";
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
+42
@@ -0,0 +1,42 @@
|
||||
#!/usr/bin/env perl
|
||||
|
||||
use strict;
|
||||
use warnings;
|
||||
|
||||
my $usage = "\n\n\tusage: $0 file.kmers bfly.log\n\n";
|
||||
|
||||
my $file_kmers = $ARGV[0] or die $usage;
|
||||
my $bfly_log = $ARGV[1] or die $usage;
|
||||
|
||||
my %kmers;
|
||||
{
|
||||
open (my $fh, $file_kmers) or die $!;
|
||||
while (<$fh>) {
|
||||
chomp;
|
||||
$kmers{$_} = 1;
|
||||
}
|
||||
close $fh;
|
||||
}
|
||||
|
||||
open (my $fh, $bfly_log) or die "Error, cannot open file $bfly_log";
|
||||
while (<$fh>) {
|
||||
my $line = $_;
|
||||
chomp;
|
||||
# EDGE_PRUNING::removeLightOutEdges() removing the edge: G:W-1(V6691_D-1) GAGGCTGTGAAGAGACTGGCAGAG -> G:W-1(V9713_D-1) GAGGCTGTGAAGAGACTGGCAGAG (weight: 1.0 <= e_edge_thr: 6.550000000000001, EDGE_THR=0.05
|
||||
|
||||
if (/^EDGE_PRUNING/) {
|
||||
my @x = split(/\s+/);
|
||||
my $kmer_A = $x[5];
|
||||
my $kmer_B = $x[6];
|
||||
|
||||
if ($kmers{$kmer_A} && $kmers{$kmer_B}) {
|
||||
print "!!\t$line";
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
exit(0);
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
#!/usr/bin/env perl
|
||||
|
||||
use strict;
|
||||
use warnings;
|
||||
use FindBin;
|
||||
use File::Basename;
|
||||
|
||||
my $usage = "\n\n\tusage: $0 target_trans_files.list [opts ex. --wgsim ...]\n\n";
|
||||
|
||||
my $target_trans_files_file = $ARGV[0] or die $usage;
|
||||
shift @ARGV;
|
||||
|
||||
|
||||
main: {
|
||||
|
||||
my @ref_files = `cat $target_trans_files_file`;
|
||||
chomp @ref_files;
|
||||
|
||||
foreach my $file (@ref_files) {
|
||||
my $outdir = dirname($file);
|
||||
my $cmd = "$FindBin::Bin/run_simulate_reads.wgsim.pl -R $file -O $outdir @ARGV";
|
||||
|
||||
print "$cmd\n";
|
||||
}
|
||||
|
||||
exit(0);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user