This commit is contained in:
2025-11-25 00:28:51 +08:00
commit eb3f16c30e
406 changed files with 91653 additions and 0 deletions
@@ -0,0 +1,46 @@
#!/usr/bin/env perl
use strict;
use warnings;
my $usage = "usage: $0 file.audit_summary\n\n";
my $filename = $ARGV[0] or die $usage;
main: {
my %yes_no_counter;
my $total_reco = 0;
my $total_ref = 0;
my $total_FL = 0;
open (my $fh, $filename) or die $!;
while (<$fh>) {
if (/ref_fa/) { next; }
chomp;
my @x = split(/\t/);
my $yes_or_no = $x[4];
$yes_no_counter{$yes_or_no}++;
my $num_reco = $x[3];
$total_reco += $num_reco;
my $num_ref = $x[1];
$total_ref += $num_ref;
my $num_FL = $x[2];
$total_FL += $num_FL;
}
close $fh;
my $num_yes = $yes_no_counter{YES} || 0;
my $num_no = $yes_no_counter{NO} ||0;
my $num_additional = $total_reco - $total_FL;
print join("\t", "#YES", "#NO", "#FL", "#TOT_Trans", "#Ref", "#additional") . "\n";
print join("\t", $num_yes, $num_no, $total_FL, $total_reco, $total_ref, $num_additional) . "\n";
exit(0);
}
@@ -0,0 +1,159 @@
#!/usr/bin/env perl
use strict;
use warnings;
use File::Basename;
use lib ($ENV{EUK_MODULES});
use Fasta_reader;
my $usage = "\n\n\tusage: $0 < list of audit.txt files from stdin \n\n\n";
my $MIN_SEQ_LEN = 1000;
my $total_genes = 0;
my $total_genes_reco = 0;
my $total_refseq_trans = 0;
my $total_iso_reco_count = 0;
my $total_extra = 0;
print join("\t", "gene_id", "reco_gene_flag", "num_refseqs", "num_FL_trin", "num_extra_trin") . "\n";
my $counter = 0;
while (<>) {
chomp;
my $dir = dirname($_);
$counter++;
my $gene_id = basename($dir);
my $trin_fasta_file = "$dir/trinity_out_dir.Trinity.fasta";
if (! -e $trin_fasta_file) {
print STDERR "ERROR: Cannot locate file: $trin_fasta_file\n";
next;
}
my $fasta_reader = new Fasta_reader($trin_fasta_file);
my %trin_seqs = $fasta_reader->retrieve_all_seqs_hash();
my %trin_lens = &get_seq_lens(%trin_seqs);
my $FL_reco_file = "$dir/FL.test.pslx.maps";
my %reco_refseq;
my %reco_trin_to_refseq = &parse_reco($FL_reco_file, \%reco_refseq);
my $num_FL_trin = scalar(keys %reco_trin_to_refseq);
my $refseqs_fa = "$dir/refseqs.fa";
my @refseq_accs = &get_accs($refseqs_fa);
my $num_refseqs = scalar(@refseq_accs);
my @failed_reco_refseqs;
foreach my $acc (@refseq_accs) {
unless ($reco_refseq{$acc}) {
push (@failed_reco_refseqs, $acc);
}
}
my $num_failed_reco_refseqs = scalar(@failed_reco_refseqs);
my @extra_trin_accs;
foreach my $trin_acc (keys %trin_seqs) {
if (! exists $reco_trin_to_refseq{$trin_acc}) {
if ($trin_lens{$trin_acc} >= $MIN_SEQ_LEN) {
push (@extra_trin_accs, $trin_acc);
}
}
}
my $num_extra_trin = scalar(@extra_trin_accs);
my $reco_gene_flag = ($num_failed_reco_refseqs == 0) ? "YES" : "NO";
print join("\t", $gene_id, $reco_gene_flag, $num_refseqs, $num_FL_trin, $num_extra_trin) . "\n";
$total_genes++;
if ($reco_gene_flag eq "YES") {
$total_genes_reco++;
}
$total_refseq_trans += $num_refseqs;
$total_iso_reco_count += $num_FL_trin;
$total_extra += $num_extra_trin;
}
if ($counter > 1) {
print "\n\n";
print join("\t", "Total_Genes", "Total_Genes_Reco", "Total_RefTrans", "Total_RefTransReco", "Total_extra_trans") . "\n";
print join("\t", $total_genes, $total_genes_reco, $total_refseq_trans, $total_iso_reco_count, $total_extra) . "\n";
}
exit(0);
####
sub get_accs {
my ($fasta_file) = @_;
my @accs;
open (my $fh, $fasta_file) or die $!;
while (<$fh>) {
if (/^>(\S+)/) {
push (@accs, $1);
}
}
return(@accs);
}
####
sub parse_reco {
my ($reco_file, $reco_refseq_href) = @_;
my %trin_to_reco_acc;
open (my $fh, $reco_file) or die $!;
while (<$fh>) {
chomp;
my ($trans_acc, $trinity_contigs) = split(/\t/);
my @trin_contigs = split(/,/, $trinity_contigs);
foreach my $trin (@trin_contigs) {
$trin_to_reco_acc{$trin}->{$trans_acc} = 1;
$reco_refseq_href->{$trans_acc} = 1;
}
}
close $fh;
return(%trin_to_reco_acc);
}
####
sub get_seq_lens {
my (%trin_seqs) = @_;
my %lens;
foreach my $acc (keys %trin_seqs) {
my $seq = $trin_seqs{$acc};
my $seqlen = length($seq);
$lens{$acc} = $seqlen;
}
return(%lens);
}
@@ -0,0 +1,46 @@
#!/usr/bin/env perl
use strict;
use warnings;
use Cwd;
use FindBin;
my $usage = "usage: $0 info_files.list.txt output_basedir [eval cmds]\n\n";
my $files_listing_file = $ARGV[0] or die $usage;
my $output_basedir = $ARGV[1] or die $usage;
shift @ARGV;
shift @ARGV;
main: {
my @files = `cat $files_listing_file`;
chomp @files;
my $eval_script = "$FindBin::Bin/run_Trinity_eval.sh";
my $basedir = cwd();
unless ($output_basedir =~ /^\//) {
$output_basedir = "$basedir/$output_basedir";
}
foreach my $file (@files) {
my $line = `cat $file`;
chomp $line;
my ($refseq_fa_file, $left_fa, $right_fa) = split(/\t/, $line);
my @pts = split(/\//, $refseq_fa_file);
my $gene_name = $pts[-2];
my $cmd = "$eval_script -R $refseq_fa_file --left $left_fa --right $right_fa -O $output_basedir/$gene_name @ARGV";
print "$cmd\n";
}
exit(0);
}
@@ -0,0 +1,308 @@
#!/usr/bin/env perl
use strict;
use warnings;
use FindBin;
use lib ("$ENV{TRINITY_HOME}/PerlLib/");
use Fasta_reader;
use Cwd;
use Data::Dumper;
use Carp;
use Getopt::Long qw(:config no_ignore_case bundling pass_through);
use List::Util qw (shuffle);
my $help_flag;
my $ref_trans_fa;
my $MIN_REFSEQ_LENGTH = 100;
my $OUT_DIR = "Seqs_dir";
my $MAX_ISOFORMS = -1;
my $MIN_ISOFORMS = 2;
my $usage = <<__EOUSAGE__;
################################################################################
#
# * Required:
#
# --ref_trans|R <string> reference transcriptome
#
# * Common Opts:
#
# --by_Gene target all isoforms of a gene at once.
# (requires multiple isoforms, ignores single-iso genes)
#
# --out_dir|O <string> output directory name (default: $OUT_DIR)
#
# * Misc Opts:
#
# --min_refseq_length <int> min length for a reference transcript
# sequence (default: $MIN_REFSEQ_LENGTH)
#
# if --by_Gene:
#
# --min_isoforms <int> default: $MIN_ISOFORMS
# --max_isoforms <int> max number of isoforms to test (default: $MAX_ISOFORMS)
#
# --longest_isoform_only restricts to only single longest isoform per gene.
#
# --restrict_to_genes <string> file containing lists of gene accessions to restrict to.
#
############################################################################################
__EOUSAGE__
;
my $BY_GENE_FLAG = 0;
my $LONGEST_ISOFORM_ONLY_FLAG = 0;
my $restrict_to_genes_file = "";
&GetOptions ( 'h' => \$help_flag,
# required
'ref_trans|R=s' => \$ref_trans_fa,
# optional
'out_dir|O=s' => \$OUT_DIR,
'min_refseq_length=i' => \$MIN_REFSEQ_LENGTH,
'by_Gene' => \$BY_GENE_FLAG,
'max_isoforms=i' => \$MAX_ISOFORMS,
'min_isoforms=i' => \$MIN_ISOFORMS,
'longest_isoform_only' => \$LONGEST_ISOFORM_ONLY_FLAG,
'restrict_to_genes=s' => \$restrict_to_genes_file,
);
if ($help_flag) {
die $usage;
}
unless ($ref_trans_fa) {
die $usage;
}
main: {
my $BASEDIR = cwd();
if ($ref_trans_fa =~ /\.gz$/) {
my $unzipped = $ref_trans_fa;
$unzipped =~ s/\.gz$//g;
if (! -s $unzipped) {
&process_cmd("gunzip -c $ref_trans_fa > $unzipped");
}
$ref_trans_fa = $unzipped;
}
my $fasta_reader = new Fasta_reader($ref_trans_fa);
my %fasta_seqs = $fasta_reader->retrieve_all_seqs_hash();
my %reorganized_fasta_seqs = &reorganize_fasta_seqs(\%fasta_seqs, $BY_GENE_FLAG);
my %restricted_genes;
if ($restrict_to_genes_file) {
my @gene_ids = `cat $restrict_to_genes_file`;
chomp @gene_ids;
%restricted_genes = map { + $_ => 1 } @gene_ids;
}
my $total_counter = 0;
my @accs = keys %reorganized_fasta_seqs;
my %seen;
foreach my $acc (@accs) {
if (%restricted_genes && ! exists $restricted_genes{$acc}) {
# skipping, not in the restricted list.
next;
}
$seen{$acc} = 1;
chdir $BASEDIR or die "Error, cannot cd to $BASEDIR";
my $seq_entries_aref = $reorganized_fasta_seqs{$acc};
my @min_length_targets;
foreach my $entry (@$seq_entries_aref) {
my ($trans_acc, $seq) = ($entry->{acc},
$entry->{seq});
if (length($seq) >= $MIN_REFSEQ_LENGTH && $seq !~ /[^GATC]/i) {
push (@min_length_targets, $entry);
}
}
unless (@min_length_targets) {
print STDERR "No min length targets to pursue for $acc .... skipping.\n";
next;
}
unless (-d $OUT_DIR) {
mkdir($OUT_DIR) or die $!;
}
@min_length_targets = reverse sort {length($a->{seq}) <=> length($b->{seq}) } @min_length_targets;
my $num_total_targets = scalar(@min_length_targets);
if ($BY_GENE_FLAG) {
if ($LONGEST_ISOFORM_ONLY_FLAG) {
@min_length_targets = shift @min_length_targets;
}
else {
if ($num_total_targets < $MIN_ISOFORMS) {
next;
}
if ($num_total_targets > $MAX_ISOFORMS) {
@min_length_targets = @min_length_targets[0..($num_total_targets-1)];
}
}
}
&prep_seqs($acc, \@min_length_targets);
$total_counter++;
if ($total_counter % 100 == 0) {
print STDERR "\n[$total_counter]\n";
}
}
if (%restricted_genes) {
# ensure we got them all
for my $seen_acc (keys %seen) {
if (exists $restricted_genes{$seen_acc}) {
delete $restricted_genes{$seen_acc};
}
}
if (%restricted_genes) {
die "Error, missing entries for restricted gene list entries: " . Dumper(\%restricted_genes);
}
else {
print STDERR "-all restricted gene entries identified and reported.\n";
}
}
print STDERR "\nDone.\n\n";
exit(0);
}
####
sub prep_seqs {
my ($acc, $entries_aref) = @_;
print STDERR "\r-processing $acc ";
my $num_entries = scalar(@$entries_aref);
my $basedir = cwd();
my $dir_tok = $acc;
$dir_tok =~ s/\W/_/g;
my $workdir = "$OUT_DIR/$dir_tok";
unless (-d $workdir) {
mkdir $workdir or die "Error, cannot mkdir $workdir";
}
my $refseqs_fa = "$workdir/refseqs.fa";
# write ref fasta seqs.
if (! -s "$refseqs_fa") {
open (my $ofh, ">$refseqs_fa") or die $!;
foreach my $entry (@$entries_aref) {
my ($entry_acc, $seq) = ($entry->{acc},
$entry->{seq});
print $ofh ">$entry_acc\n$seq\n";
}
close $ofh;
}
return;
}
####
sub process_cmd {
my ($cmd) = @_;
print STDERR "CMD: $cmd\n";
my $ret = system($cmd);
if ($ret) {
die "Error, cmd: $cmd died with ret $ret";
}
return;
}
####
sub reorganize_fasta_seqs {
my ($fasta_seqs_href, $by_gene_flag) = @_;
my %reorg_fasta;
foreach my $acc (sort keys %$fasta_seqs_href) {
my $seq = uc $fasta_seqs_href->{$acc};
my $key = $acc;
if ($by_gene_flag) {
if ($acc =~ /^([^;]+);([^;]+)$/) {
my $trans = $1;
my $gene = $2;
$key = $gene;
}
elsif ($acc =~ /^([^\|]+)\|([^\|]+)$/) {
my $gene = $1;
my $trans = $2;
$key = $gene;
}
else {
confess "Error, no gene ID extracted from $acc ";
}
}
push (@{$reorg_fasta{$key}}, { acc => $acc,
seq => $seq}
);
}
return(%reorg_fasta);
}
@@ -0,0 +1,461 @@
#!/usr/bin/env perl
use strict;
use warnings;
use FindBin;
use lib ("$ENV{TRINITY_HOME}/PerlLib/");
use Fasta_reader;
use Cwd;
use Data::Dumper;
use Carp;
use Getopt::Long qw(:config no_ignore_case bundling pass_through);
use List::Util qw (shuffle);
my $VERBOSITY_LEVEL = 10;
my $help_flag;
my $ref_trans_fa;
my $BFLY_JAR = "$ENV{TRINITY_HOME}/Butterfly/Butterfly.jar";
my $INCLUDE_REF_TRANS = 0;
my $OUT_DIR = "testing_dir";
my $MIN_CONTIG_LENGTH = 200;
my $min_per_id = 90;
my $usage = <<__EOUSAGE__;
################################################################################
#
# * Required:
#
# --ref_trans|R <string> reference transcriptome
#
# --left <string> left reads fa file
#
# --right <string> right reads fa file
#
# * Common Opts:
#
# --bfly_jar|B <string> Butterfly jar file
#
# --out_dir|O <string> output directory name (default: $OUT_DIR)
#
# * include FL seq opts:
#
# --incl_ref_trans include the ref transcript as a long
# read (default: off)
#
# * Misc Opts:
#
# --incl_ref_dot include dot files for the reference sequences
#
# -V <int> verbosity level (default: 12)
#
# --acc <string> restrict to a specific accession (gene or transcript)
#
# --paired_as_single treat paired reads as single reads
#
# --min_contig_length <int> minimum contig length for Trinity assembly. default: $MIN_CONTIG_LENGTH
#
# --strict weld all, no pruning or path merging.
#
# --no_cleanup no cleaning up of trinity output.
#
# --strand_specific sets to strand-specific mode (RF)
#
# --bfly_opts <string> butterfly additional opts
#
############################################################################################
__EOUSAGE__
;
my $NO_CLEANUP = 0;
my $INCLUDE_REF_DOT_FILES = 0;
my $PAIRED_AS_SINGLE = "";
my $SHUFFLE = 0;
my $MAX_ISOFORMS = -1;
my $STRICT = 0;
my $strand_specific_flag = 0;
my $left_file = "";
my $right_file = "";
my $BFLY_OPTS;
&GetOptions ( 'h' => \$help_flag,
# required
'ref_trans|R=s' => \$ref_trans_fa,
'left=s' => \$left_file,
'right=s' => \$right_file,
# optional
'out_dir|O=s' => \$OUT_DIR,
'bfly_jar|B=s' => \$BFLY_JAR,
'incl_ref_trans' => \$INCLUDE_REF_TRANS,
'strict' => \$STRICT,
'V=i' => \$VERBOSITY_LEVEL,
'incl_ref_dot' => \$INCLUDE_REF_DOT_FILES,
'paired_as_single' => \$PAIRED_AS_SINGLE,
'min_contig_length=i' => \$MIN_CONTIG_LENGTH,
'no_cleanup' => \$NO_CLEANUP,
'strand_specific' => \$strand_specific_flag,
'min_per_id=i' => \$min_per_id,
'bfly_opts=s' => \$BFLY_OPTS,
);
if ($help_flag) {
die $usage;
}
unless ($ref_trans_fa && $BFLY_JAR && $left_file && $right_file) {
die $usage;
}
$NO_CLEANUP = 1; ## NEEDED NOW for iworm and bfy pruning assessment
unless ($ENV{TRINITY_HOME}) {
$ENV{TRINITY_HOME} = "$FindBin::Bin/../../trinityrnaseq/";
}
my $reconstructions_log_file = "$OUT_DIR.reconstruction_summary.txt";
if ($PAIRED_AS_SINGLE) {
$PAIRED_AS_SINGLE = "--TREAT_PAIRS_AS_SINGLE";
}
main: {
my $BASEDIR = cwd();
unless ($ref_trans_fa =~ /^\//) {
$ref_trans_fa = "$BASEDIR/$ref_trans_fa";
}
unless (-d $OUT_DIR) {
&process_cmd("mkdir -p $OUT_DIR");
}
chdir $OUT_DIR or die "Error, cannot cd to $OUT_DIR";
my ($num_reco_FL, $num_ref_entries, $num_trans_reco,
$has_all_iworm_kmers, $has_all_precious_edges,
$num_LR_threaded) = &execute_seq_pipe($ref_trans_fa, $left_file, $right_file);
# num_FL: number of transcripts reconstructed as full-length
# num_entries: number of reference isoforms
# num_transcripts: total number of Trinity transcripts reconstructed.
open (my $ofh, ">audit.txt") or die $!;
my $captured_all = ($num_reco_FL == $num_ref_entries) ? "YES" : "NO";
my $header = join("\t", "ref_fa", "num_ref", "num_FL", "num_reco", "captured_all", "iworm_ok", "bfly_edges_ok",
"num_LR_threaded", "all_LR_threaded_ok");
my $all_LR_threaded_ok = ($num_LR_threaded == $num_ref_entries) ? "YES" : "NO";
my $summary = join("\t", $ref_trans_fa,
$num_ref_entries,
$num_reco_FL,
$num_trans_reco,
$captured_all,
$has_all_iworm_kmers,
$has_all_precious_edges,
$num_LR_threaded,
$all_LR_threaded_ok);
$summary = "$header\n$summary\n";
print $ofh $summary;
print $summary;
close $ofh;
&process_cmd("echo " . cwd() . "/audit.txt | $FindBin::Bin/audit_summary_stats.reexamine.pl | tee audit2.txt");
exit(0);
}
####
sub execute_seq_pipe {
my ($ref_trans_fa, $left_file, $right_file) = @_;
my $cmd = "ln -sf $ref_trans_fa $left_file $right_file .";
&process_cmd($cmd);
if (-d "trinity_out_dir") {
`rm -rf ./trinity_out_dir`;
}
my $num_entries = `grep '>' $ref_trans_fa | wc -l `;
$num_entries =~ /(\d+)/ or die "Error, cannot parse number of entries from $ref_trans_fa";
$num_entries = $1;
# run Trinity
my $bfly_jar_txt = "";
if ($BFLY_JAR) {
$bfly_jar_txt = " --bfly_jar $BFLY_JAR ";
}
$cmd = "set -o pipefail; $ENV{TRINITY_HOME}/Trinity --seqType fa --max_memory 4G --bflyHeapSpaceMax 10G --max_reads_per_graph 10000000 --group_pairs_distance 10000 --verbose_level 2 --CPU 1 ";
if ($NO_CLEANUP) {
$cmd .= " --no_cleanup ";
}
else {
$cmd .= " --full_cleanup ";
}
if ($INCLUDE_REF_TRANS) {
$cmd .= " --long_reads $ref_trans_fa ";
}
$cmd .= " --left $left_file --right $right_file ";
if ($strand_specific_flag) {
$cmd .= " --SS_lib_type RF ";
}
$cmd .= " --CPU 2 $bfly_jar_txt --inchworm_cpu 1 --min_contig_length $MIN_CONTIG_LENGTH --trinity_complete @ARGV";
my $bfly_opts = " --bfly_opts \"--generate_intermediate_dot_files -R 1 --generate_intermediate_dot_files $PAIRED_AS_SINGLE --stderr -V $VERBOSITY_LEVEL $BFLY_OPTS\" ";
if ($STRICT) {
$cmd .= " --no_bowtie --chrysalis_debug_weld_all "
. " --iworm_opts \"--no_prune_error_kmers --min_assembly_coverage 1 --min_seed_entropy 0 --min_seed_coverage 1 \" ";
$bfly_opts = " --bfly_opts \"--dont-collapse-snps --no_pruning --no_path_merging --no_remove_lower_ranked_paths --NO_EM_REDUCE --MAX_READ_SEQ_DIVERGENCE=0 --NO_DP_READ_TO_VERTEX_ALIGN --generate_intermediate_dot_files -R 1 -F 100000 --generate_intermediate_dot_files $PAIRED_AS_SINGLE --stderr -V $VERBOSITY_LEVEL $BFLY_OPTS\" ";
}
else {
#$cmd .= " --no_bowtie --chrysalis_debug_weld_all ";
#$cmd .= " --iworm_opts \" --min_seed_entropy 1 \" --min_glue 1 ";
}
$cmd .= " $bfly_opts 2>&1 | tee trin.log";
{
open (my $ofh, ">runTrinity.cmd") or die $!;
print $ofh $cmd;
close $ofh;
}
&process_cmd($cmd);
if ($NO_CLEANUP) {
rename("trinity_out_dir/Trinity.fasta", "trinity_out_dir.Trinity.fasta");
}
## check inchworm kmer content of reference sequences
&process_cmd("$ENV{TRINITY_HOME}/util/misc/print_kmers.pl $ref_trans_fa 24 > ref_kmers");
my $iworm_file = (-s "trinity_out_dir/inchworm.K25.L25.fa") ? "trinity_out_dir/inchworm.K25.L25.fa" : "trinity_out_dir/inchworm.K25.L25.DS.fa";
my $has_all_iworm_kmers = &check_inchworm_kmer_content($ref_trans_fa, $iworm_file);
## examine pruning of precious edges
my ($has_all_precious_edges) = &check_pruning("trin.log", "ref_kmers");
## see if all LR are threaded through the graph
my $num_LR_threaded = &count_num_LR_threaded("trin.log");
if ($INCLUDE_REF_DOT_FILES) {
# generate sequence graphs just refseqs
$cmd = "$ENV{TRINITY_HOME}/util/misc/Monarch --misc_seqs $ref_trans_fa --graph refseqs.dot";
if (! -s "refseqs.dot") {
&process_cmd($cmd);
}
# generate sequence graphs just refseqs
$cmd = "$ENV{TRINITY_HOME}/util/misc/Monarch --misc_seqs $ref_trans_fa,trinity_out_dir/inchworm.K25.L25.fa --graph refseqs_w_iworm.dot";
if (! -s "refseqs_w_iworm.dot") {
&process_cmd($cmd);
}
# generate sequence graphs combining all
$cmd = "$ENV{TRINITY_HOME}/util/misc/Monarch --misc_seqs $ref_trans_fa,trinity_out_dir/inchworm.K25.L25.fa,trinity_out_dir.Trinity.fasta --graph all_compare.dot";
if (! -s "all_compare.dot") {
&process_cmd($cmd);
}
}
# compare refseqs to the trinity assemblies
$cmd = "$ENV{TRINITY_HOME}/util/misc/illustrate_ref_comparison.pl $ref_trans_fa trinity_out_dir.Trinity.fasta $min_per_id | tee ref_compare.ascii_illus";
&process_cmd($cmd);
## get number of transcripts reconstructed:
$cmd = "grep '>' trinity_out_dir.Trinity.fasta | wc -l";
my $result = `$cmd`;
$result =~ s/^\s+//g;
my ($num_transcripts, @rest) = split(/\s+/, $result);
# reconstruction test
$cmd = "$ENV{TRINITY_HOME}/Analysis/FL_reconstruction_analysis/FL_trans_analysis_pipeline.pl --target $ref_trans_fa --query trinity_out_dir.Trinity.fasta --no_reuse --out_prefix FL.test --allow_non_unique_mappings --min_per_length 90 --min_per_id $min_per_id | tee FL_analysis.txt";
&process_cmd("echo $cmd > FL.cmd");
my @results = `$cmd`;
print @results;
chomp @results;
shift @results;
shift @results;
$result = shift @results;
$result =~ s/^\s+//;
my @pts = split(/\s+/, $result);
my $num_FL = $pts[2] || 0;
my $got_all_flag = 0;
if ($num_FL == $num_entries) {
print STDERR "-got all FL ($num_FL reconstructed / $num_entries total reconstructed).\n";
$got_all_flag = 1;
#print STDERR Dumper(\@pts);
}
else {
print STDERR "** missed at least one reconstructed isoform ($num_FL reconstructed / $num_entries total reconstructed).\n";
}
# cleanup really needed after all.
unless ($NO_CLEANUP) {
system("rm -rf ./trinity_out_dir");
}
return ($num_FL, $num_entries, $num_transcripts, $has_all_iworm_kmers, $has_all_precious_edges, $num_LR_threaded);
}
####
sub process_cmd {
my ($cmd) = @_;
print STDERR "CMD: $cmd\n";
my $ret = system($cmd);
if ($ret) {
die "Error, cmd: $cmd died with ret $ret";
}
return;
}
####
sub check_inchworm_kmer_content {
my ($ref_trans_fa, $iworm_file) = @_;
my $cmd = "$ENV{TRINITY_HOME}/Inchworm/bin/inchworm --threadFasta $ref_trans_fa --reads $iworm_file > $iworm_file.ref_kmer_check";
&process_cmd($cmd);
my $has_all_kmers = "YES";
open (my $fh, "$iworm_file.ref_kmer_check") or die $!;
while (<$fh>) {
chomp;
my @x = split(/\t/);
my $kmer_info = $x[1];
if ($kmer_info && $kmer_info =~ /:/) {
my ($kmer, $count) = split(/:/, $kmer_info);
if ($count == 0) {
print "Inchworm missing kmer: $kmer\n";
$has_all_kmers = "NO";
}
}
}
close $fh;
return($has_all_kmers);
}
####
sub check_pruning {
my ($log_file, $ref_kmers_file) = @_;
my $pruned_ref_edges = `$FindBin::Bin/util/find_pruned_edges_shouldve_kept.pl $ref_kmers_file $log_file`;
if ($pruned_ref_edges =~ /\w/) {
print "Pruned precious edges: $pruned_ref_edges\n";
return("NO");
}
else {
print "no pruning of precious edges\n";
return("YES");
}
}
####
sub count_num_LR_threaded {
my ($logfile) = @_;
# FINAL BEST PATH for LR$|ENST00000479454.1;ASZ1_mutated is [1, 228, 373, 2906, 598, 2885, 771] with total mm: 55
# No read mapping found for: LR$|ENST00000465832.1;ASZ1_mutated
my $num_LR_threaded = 0;
open(my $fh, $logfile) or die "Error, cannot open file $logfile";
while(<$fh>) {
chomp;
if (/FINAL BEST PATH for LR\$/) {
print "$_\n";
$num_LR_threaded++;
}
elsif (/No read mapping found for: LR\$/) {
print "$_\n";
}
}
close $fh;
return($num_LR_threaded);
}
@@ -0,0 +1,16 @@
#!/bin/bash
#source /broad/software/scripts/useuse
#reuse Perl-5.8
#reuse .samtools-0.1.19
#reuse GCC-4.9
CMD="`dirname $0`/run_Trinity_eval.pl $*"
eval $CMD
exit $?
@@ -0,0 +1,148 @@
#!/usr/bin/env perl
use strict;
use warnings;
use FindBin;
use lib ("$ENV{TRINITY_HOME}/PerlLib/");
use Fasta_reader;
use Cwd;
use Data::Dumper;
use Carp;
use Getopt::Long qw(:config no_ignore_case bundling pass_through);
use List::Util qw (shuffle);
my $help_flag;
my $usage = <<__EOUSAGE__;
################################################################################
$0
################################################################################
#
# * Required:
#
# --ref_trans|R <string> reference transcriptome
#
# --out_dir|O <string> output directory name
#
# --read_length <int> default: 76
#
# --frag_length <int> default: 300
#
# --depth_of_cov <int> default: 100
#
#
####
#
# following wgsim options are pass-through:
#
# Options:
# -e FLOAT base error rate [0.020]
# -s INT standard deviation [50]
# -r FLOAT rate of mutations [0.0010]
# -R FLOAT fraction of indels [0.15]
# -X FLOAT probability an indel is extended [0.30]
# -S INT seed for random generator [-1]
# -A FLOAT disgard if the fraction of ambiguous bases higher than FLOAT [0.05]
# -h haplotype mode
# -Z INT strand specific mode: 1=FR, 2=RF
# -D debug mode... highly verbose
#
#
############################################################################################
__EOUSAGE__
;
my $OUT_DIR;
my $ref_trans_fa;
my $read_length = 76;
my $frag_length = 300;
my $depth_of_cov = 100;
&GetOptions ( 'help' => \$help_flag,
# required
'ref_trans|R=s' => \$ref_trans_fa,
# optional
'out_dir|O=s' => \$OUT_DIR,
'read_length=i' => \$read_length,
'frag_length=i' => \$frag_length,
'depth_of_cov=i' => \$depth_of_cov,
);
if ($help_flag) {
die $usage;
}
unless ($ref_trans_fa && $OUT_DIR) {
die $usage;
}
unless ($ENV{TRINITY_HOME}) {
$ENV{TRINITY_HOME} = "$FindBin::Bin/../../trinityrnaseq/";
}
main: {
my $BASEDIR = cwd();
unless ($ref_trans_fa =~ /^\//) {
$ref_trans_fa = "$BASEDIR/$ref_trans_fa";
}
unless (-d $OUT_DIR) {
&process_cmd("mkdir -p $OUT_DIR");
}
chdir $OUT_DIR or die "Error, cannot cd to $OUT_DIR";
my $cmd = "";
# simulate reads:
$cmd = "$ENV{TRINITY_HOME}/util/misc/simulate_illuminaPE_from_transcripts.wgsim.pl --transcripts $ref_trans_fa "
. " --read_length $read_length "
. " --frag_length $frag_length "
. " --depth_of_cov 200 "
. " @ARGV "; # wgsim opts pass-through
;
## todo: add mutation rate info
&process_cmd($cmd);
}
####
sub process_cmd {
my ($cmd) = @_;
print STDERR "CMD: $cmd\n";
my $ret = system($cmd);
if ($ret) {
die "Error, cmd: $cmd died with ret $ret";
}
return;
}
@@ -0,0 +1,42 @@
#!/usr/bin/env perl
use strict;
use warnings;
my $usage = "\n\n\tusage: $0 file.kmers bfly.log\n\n";
my $file_kmers = $ARGV[0] or die $usage;
my $bfly_log = $ARGV[1] or die $usage;
my %kmers;
{
open (my $fh, $file_kmers) or die $!;
while (<$fh>) {
chomp;
$kmers{$_} = 1;
}
close $fh;
}
open (my $fh, $bfly_log) or die "Error, cannot open file $bfly_log";
while (<$fh>) {
my $line = $_;
chomp;
# EDGE_PRUNING::removeLightOutEdges() removing the edge: G:W-1(V6691_D-1) GAGGCTGTGAAGAGACTGGCAGAG -> G:W-1(V9713_D-1) GAGGCTGTGAAGAGACTGGCAGAG (weight: 1.0 <= e_edge_thr: 6.550000000000001, EDGE_THR=0.05
if (/^EDGE_PRUNING/) {
my @x = split(/\s+/);
my $kmer_A = $x[5];
my $kmer_B = $x[6];
if ($kmers{$kmer_A} && $kmers{$kmer_B}) {
print "!!\t$line";
}
}
}
exit(0);
@@ -0,0 +1,28 @@
#!/usr/bin/env perl
use strict;
use warnings;
use FindBin;
use File::Basename;
my $usage = "\n\n\tusage: $0 target_trans_files.list [opts ex. --wgsim ...]\n\n";
my $target_trans_files_file = $ARGV[0] or die $usage;
shift @ARGV;
main: {
my @ref_files = `cat $target_trans_files_file`;
chomp @ref_files;
foreach my $file (@ref_files) {
my $outdir = dirname($file);
my $cmd = "$FindBin::Bin/run_simulate_reads.wgsim.pl -R $file -O $outdir @ARGV";
print "$cmd\n";
}
exit(0);
}