commit eb3f16c30ea804f286e1b60867f603e198a30f8d Author: IvisTang Date: Tue Nov 25 00:28:51 2025 +0800 20251125 diff --git a/.envrc b/.envrc new file mode 100644 index 0000000..596826f --- /dev/null +++ b/.envrc @@ -0,0 +1,8 @@ +set +u +export PROJECTHOME="/home/ywtang/project/biyelunwen" +export SCRIPTS="$PROJECTHOME/99.scripts" +export PUEUE_CONFIG_PATH="$PROJECTHOME/.pueue.yml" +watch_file pixi.lock +eval "$(pixi shell-hook)" +PATH_add $SCRIPTS/bucky/bin +PATH_add $SCRIPTS/ticr diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..887a2c1 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,2 @@ +# SCM syntax highlighting & preventing 3-way merges +pixi.lock merge=binary linguist-language=YAML linguist-generated=true diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..93d9614 --- /dev/null +++ b/.gitignore @@ -0,0 +1,16 @@ +# pixi environments +.pixi/* +!.pixi/config.toml +.pueue/* +00.rawdata/* +01.reference/* +02.mapping_assembly/* +03.denovo_assembly/* +04.plastid/* +05.reduce_redundancy/* +06.phylogeny_reconstruction/* +98.results/* +99.scripts/bucky/ +99.scripts/phyparts/ +.pueue.yml +run.status diff --git a/.vscode/settings.json b/.vscode/settings.json new file mode 100644 index 0000000..fe57298 --- /dev/null +++ b/.vscode/settings.json @@ -0,0 +1,5 @@ +{ + "julia.executablePath": "/home/ywtang/project/biyelunwen/.pixi/envs/default/bin/julia", + "julia.enableCrashReporter": false, + "julia.enableTelemetry": false +} \ No newline at end of file diff --git a/99.scripts/bucky b/99.scripts/bucky new file mode 160000 index 0000000..9228d7c --- /dev/null +++ b/99.scripts/bucky @@ -0,0 +1 @@ +Subproject commit 9228d7c9446e619df6fd609c3f48d74472b50c96 diff --git a/99.scripts/miscs/__pycache__/hmmsearch_result_to_new_fasta.cpython-311.pyc b/99.scripts/miscs/__pycache__/hmmsearch_result_to_new_fasta.cpython-311.pyc new file mode 100644 index 0000000..c972e90 Binary files /dev/null and b/99.scripts/miscs/__pycache__/hmmsearch_result_to_new_fasta.cpython-311.pyc differ diff --git a/99.scripts/miscs/check_and_translate_outgroup_cds.py b/99.scripts/miscs/check_and_translate_outgroup_cds.py new file mode 100755 index 0000000..bde050a --- /dev/null +++ b/99.scripts/miscs/check_and_translate_outgroup_cds.py @@ -0,0 +1,144 @@ +#! /usr/bin/env python3 +#!/usr/bin/env python3 +""" +CDS to Protein Converter with Internal Stop Codon Filtering + +This script processes CDS sequences from a FASTA file, translates them to protein sequences, +checks for internal stop codons, and outputs clean CDS and protein sequences. +""" + +import sys +from Bio import SeqIO +from Bio.SeqRecord import SeqRecord + + +def translate_cds_and_filter( + input_fasta, output_clean_cds, output_clean_protein, translation_table=1 +): + """ + Main processing function: Translates CDS sequences and filters those with internal stop codons[2](@ref) + + Parameters: + input_fasta: Path to input CDS sequences FASTA file + output_clean_cds: Path for output clean CDS sequences + output_clean_protein: Path for output protein sequences + translation_table: Genetic code table number (default: 1 = Standard) + + Returns: + Tuple of (clean_cds_count, removed_count) + """ + clean_cds_records = [] # Store CDS sequences without internal stop codons + clean_protein_records = [] # Store corresponding protein sequences + removed_count = 0 # Count of removed sequences + total_count = 0 # Total sequences processed + + print(f"Processing file: {input_fasta}") + + # Process each sequence in the input FASTA file + for record in SeqIO.parse(input_fasta, "fasta"): + total_count += 1 + cds_seq = record.seq + seq_id = record.id + + # Check if sequence length is multiple of 3 + if len(cds_seq) % 3 != 0: + print(f"Warning: Sequence {seq_id} length is not multiple of 3, skipping.") + removed_count += 1 + continue + + try: + # Translate CDS to protein sequence (including stop codon '*') + protein_seq = cds_seq.translate(table=translation_table, to_stop=False) + protein_str = str(protein_seq) + + # Find all stop codon positions in the protein sequence + stop_positions = [i for i, aa in enumerate(protein_str) if aa == "*"] + has_internal_stop = False + + # Check if any stop codon is not at the end (internal stop) + if stop_positions: + last_position = len(protein_str) - 1 + # Internal stop exists if stop codon is found not at the very end + if any(pos != last_position for pos in stop_positions): + has_internal_stop = True + + if has_internal_stop: + # Skip sequences with internal stop codons + print( + f"Warning: Removing sequence {seq_id}: Internal stop codon detected" + ) + removed_count += 1 + else: + # Create clean protein sequence (remove terminal stop codon if present) + if protein_str.endswith("*"): + protein_seq_clean = protein_seq[:-1] # Remove terminal stop codon + else: + protein_seq_clean = protein_seq + + # Create protein sequence record + protein_record = SeqRecord( + seq=protein_seq_clean, id=seq_id, description=record.description + ) + + # Add to results + clean_cds_records.append(record) + clean_protein_records.append(protein_record) + + except Exception as e: + print(f"Error processing sequence {seq_id}: {e}") + removed_count += 1 + continue + + # Write output files if we have valid sequences + if clean_cds_records: + SeqIO.write(clean_cds_records, output_clean_cds, "fasta") + SeqIO.write(clean_protein_records, output_clean_protein, "fasta") + + print("\nProcessing completed successfully!") + print(f"Total input sequences: {total_count}") + print(f"Sequences retained: {len(clean_cds_records)}") + print(f"Sequences removed: {removed_count}") + print(f"Clean CDS sequences saved to: {output_clean_cds}") + print(f"Protein sequences saved to: {output_clean_protein}") + + return len(clean_cds_records), removed_count + else: + print("Warning: No sequences passed filtering. Please check input file format.") + return 0, removed_count + + +def main(): + """Main command-line interface function""" + if len(sys.argv) != 3: + print("Usage: python cds_to_protein_filter.py input.fasta output_stem") + print("Arguments:") + print(" input.fasta Input CDS sequences FASTA file") + print(" output_stem Stem for output files ") + sys.exit(1) + + input_file = sys.argv[1] + output_stem = sys.argv[2] + output_cds_file = f"{output_stem}.cds.fa" + output_protein_file = f"{output_stem}.pep.fa" + + # Verify input file exists + try: + with open(input_file, "r"): + pass + except FileNotFoundError: + print(f"Error: Input file {input_file} not found!") + sys.exit(1) + except IOError as e: + print(f"Error reading input file {input_file}: {e}") + sys.exit(1) + + # Execute processing + try: + translate_cds_and_filter(input_file, output_cds_file, output_protein_file) + except Exception as e: + print(f"Fatal error during processing: {e}") + sys.exit(1) + + +if __name__ == "__main__": + main() diff --git a/99.scripts/miscs/hmmsearch_result_to_new_ogs_list.py b/99.scripts/miscs/hmmsearch_result_to_new_ogs_list.py new file mode 100755 index 0000000..30d1bf8 --- /dev/null +++ b/99.scripts/miscs/hmmsearch_result_to_new_ogs_list.py @@ -0,0 +1,215 @@ +#! /usr/bin/env python3 +import os +import re +import argparse +from Bio import SeqIO +from pathlib import Path + + +def parse_fasta(fasta_file_path): + """ + Parse a FASTA file and return a list of sequence ids. + Args: + fasta_file_path: Path to the FASTA file + Returns: + list: List of sequence ids + """ + sequence_ids = [] + try: + for record in SeqIO.parse(fasta_file_path, "fasta"): + sequence_ids.append(record.id) + except Exception as e: + print(f"Error parsing FASTA file {fasta_file_path}: {e}") + return sequence_ids + + +def parse_hmmer_tbl(tbl_file_path): + """ + Parse HMMER tbl format result file and extract best hit information + + Args: + tbl_file_path: Path to the tbl file + + Returns: + dict: Best hit information + """ + best_hit = {} + + try: + with open(tbl_file_path, "r") as f: + for line in f: + # Skip comment lines + if line.startswith("#"): + continue + + # Split line (tbl format is typically space or tab separated) + parts = re.split(r"\s+", line.strip()) + if len(parts) < 5: + continue + + # Extract key information: target name, E-value, score, etc. + # tbl format columns: target name, target accession, query name, query accession, E-value, score, etc. + target_name = parts[0] + e_value = float(parts[4]) # Full sequence E-value + + # If it's a new sequence or we found a better hit (lower E-value) + if not best_hit.keys() or e_value < best_hit["e_value"]: + best_hit = { + "e_value": e_value, + "target_name": target_name, + } + + except Exception as e: + print(f"Error parsing tbl file {tbl_file_path}: {e}") + return {} + + return best_hit + + +def find_corresponding_files(fasta_dir, tbl_dir): + """ + Find corresponding FASTA and tbl file pairs + + Args: + fasta_dir: Directory containing FASTA files + tbl_dir: Directory containing tbl result files + + Returns: + list: List of (fasta_file_path, tbl_file_path) tuples + """ + file_pairs = [] + + # Get all FASTA files + fasta_files = {} + + for fasta_path in Path(fasta_dir).glob("*.fa"): + stem = fasta_path.stem + fasta_files[stem] = fasta_path + + # Find corresponding tbl files + for stem, fasta_path in fasta_files.items(): + tbl_name = f"{stem}.fa.hmmsearch.tblout" + tbl_path = Path(tbl_dir) / tbl_name + if tbl_path.exists(): + file_pairs.append((fasta_path, tbl_path)) + else: + print(f"Warning: No corresponding tbl file found for {stem}") + + return file_pairs + + +def add_best_hit_to_og_list(file_pair): + """ + Add best hit information to FASTA file + + Args: + file_pair: Tuple of (fasta_path, tbl_path) + Returns: + tuple: (stem, seq_ids) + """ + fasta_path, tbl_path = file_pair + + # Parse tbl file to get best hits + best_hit = parse_hmmer_tbl(tbl_path) + + if not best_hit: + print(f"Warning: No valid hits found in {tbl_path}") + return fasta_path.stem, [] + + best_seq = best_hit["target_name"] + seq_ids = parse_fasta(fasta_path) + seq_ids.append(best_seq) + + return fasta_path.stem, seq_ids + + +def process_all_files(fasta_dir, tbl_dir, output_file): + """ + Process all FASTA and tbl file pairs + + Args: + fasta_dir: Directory containing FASTA files + tbl_dir: Directory containing tbl result files + output_file: Output file path + """ + print("Starting file processing...") + print(f"FASTA directory: {fasta_dir}") + print(f"tbl directory: {tbl_dir}") + print(f"Output file: {output_file}") + print("-" * 50) + + og_list = {} + + # Find corresponding file pairs + file_pairs = find_corresponding_files(fasta_dir, tbl_dir) + + if not file_pairs: + print("No corresponding FASTA-tbl file pairs found") + return + + print(f"Found {len(file_pairs)} file pairs to process") + + # Process each file pair + for i, file_pair in enumerate(file_pairs, 1): + print(f"\nProcessing pair {i}/{len(file_pairs)}:") + stem, seq_ids = add_best_hit_to_og_list(file_pair) + if not seq_ids: + print(f"Skipping {stem} due to no valid hits") + continue + og_list[stem] = seq_ids + + with open(output_file, "w") as out_f: + for og, ids in og_list.items(): + out_f.write(f"{og}\t" + "\t".join(ids) + "\n") + + print(f"\nProcessing completed! All results saved to: {output_file}") + + +def main(fasta_directory, tbl_directory, output_file): + """ + Main function - take paths and run processing + + Args: + fasta_directory: Folder containing FASTA files + tbl_directory: Folder containing tbl result files + output_file: Output file path + """ + # Check if input directories exist + if not os.path.exists(fasta_directory): + print(f"Error: FASTA directory does not exist {fasta_directory}") + return + + if not os.path.exists(tbl_directory): + print(f"Error: tbl directory does not exist {tbl_directory}") + return + + # Process all files + process_all_files(fasta_directory, tbl_directory, output_file) + + +if __name__ == "__main__": + # Parse command-line arguments and call main + parser = argparse.ArgumentParser( + description="Add HMMER best hit into orthologs FASTA." + ) + parser.add_argument( + "-f", + "--fasta_directory", + required=True, + help="Directory containing FASTA files", + ) + parser.add_argument( + "-t", + "--tbl_directory", + required=True, + help="Directory containing tbl result files", + ) + parser.add_argument( + "-o", + "--output_file", + required=True, + help="File to write output seqname", + ) + args = parser.parse_args() + + main(args.fasta_directory, args.tbl_directory, args.output_file) diff --git a/99.scripts/miscs/ogs_list_to_ogs_fasta.py b/99.scripts/miscs/ogs_list_to_ogs_fasta.py new file mode 100755 index 0000000..5c5e0ef --- /dev/null +++ b/99.scripts/miscs/ogs_list_to_ogs_fasta.py @@ -0,0 +1,92 @@ +#! /usr/bin/env python3 +import os +import argparse +from Bio import SeqIO +from Bio.SeqRecord import SeqRecord +from pathlib import Path + + +def parse_fasta(fasta_file_path): + """ + Parse a FASTA file and return a list of SeqRecord. + Args: + fasta_file_path: Path to the FASTA file + Returns: + list: List of SeqRecord + """ + try: + records = list(SeqIO.parse(fasta_file_path, "fasta")) + return records + except Exception as e: + print(f"Error parsing FASTA file {fasta_file_path}: {e}") + return [] + + +def ogs_to_fasta(ogs_name, seq_list, source_records, output_dir): + """ + Convert OGS list to FASTA format. + + Args: + ogs_name: Name of the OGS + seq_list: List of sequence IDs + source_records: List of SeqRecord from source FASTA + output_dir: Directory to save the output FASTA file + """ + output_path = Path(output_dir) / f"{ogs_name}.fa" + seq_dict = {record.id: record for record in source_records} + + with open(output_path, "w") as out_f: + for seq_id in seq_list: + if seq_id in seq_dict: + updated_id = seq_id.split("@")[0] + updated_record = SeqRecord( + seq_dict[seq_id].seq, + id=updated_id, + description="", + ) + SeqIO.write(updated_record, out_f, "fasta") + else: + print(f"Warning: Sequence ID {seq_id} not found in source records.") + + +def process_ogs_list(ogs_file, source_fasta, output_dir): + """ + Process OGS list file and convert each OGS to FASTA format. + + Args: + ogs_file: Path to the OGS list file + source_fasta: Path to the source FASTA file + output_dir: Directory to save the output FASTA files + """ + source_records = parse_fasta(source_fasta) + + with open(ogs_file, "r") as f: + for line in f: + parts = line.strip().split("\t") + if len(parts) < 2: + continue + ogs_name = parts.pop(0) + ogs_to_fasta(ogs_name, parts, source_records, output_dir) + print(f"Processed OGS: {ogs_name}") + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description="Convert OGS list to FASTA format.") + parser.add_argument( + "-i", "--input_ogs", required=True, help="Path to the OGS list file" + ) + parser.add_argument( + "-s", "--source_fasta", required=True, help="Path to the source FASTA file" + ) + parser.add_argument( + "-o", + "--output_dir", + required=True, + help="Directory to save the output FASTA files", + ) + + args = parser.parse_args() + + os.makedirs(args.output_dir, exist_ok=True) + + process_ogs_list(args.input_ogs, args.source_fasta, args.output_dir) diff --git a/99.scripts/miscs/phypartspiecharts.py b/99.scripts/miscs/phypartspiecharts.py new file mode 100755 index 0000000..04b2775 --- /dev/null +++ b/99.scripts/miscs/phypartspiecharts.py @@ -0,0 +1,226 @@ +#!/usr/bin/env python + +from xvfbwrapper import Xvfb +import argparse +import re +import json +from ete3 import Tree, TreeStyle, NodeStyle, faces + +helptext= ''' +Generate the "Pie Chart" representation of gene tree conflict from Smith et al. 2015 from +the output of phyparts, the bipartition summary software described in the same paper. + +The input files include three files produced by PhyParts, and a file containing a species +tree in Newick format (likely, the tree used for PhyParts). The output is an SVG containing +the phylogeny along with pie charts at each node. + +Requirements: + +Python 3 +ete3 + +''' + + + +vdisplay = Xvfb() +vdisplay.start() + + + + +#Read in species tree and convert to ultrametric + +#Match phyparts nodes to ete3 nodes +def get_phyparts_nodes(sptree_fn,phyparts_root): + sptree = Tree(sptree_fn) + sptree.convert_to_ultrametric() + + phyparts_node_key = [line for line in open(phyparts_root+".node.key")] + subtrees_dict = {n.split()[0]:Tree(n.split()[1]+";") for n in phyparts_node_key} + subtrees_topids = {} + for x in subtrees_dict: + subtrees_topids[x] = subtrees_dict[x].get_topology_id() + #print(subtrees_topids['1']) + #print() + for node in sptree.traverse(): + node_topid = node.get_topology_id() + if "Takakia_4343a" in node.get_leaf_names(): + print(node_topid) + print(node) + for subtree in subtrees_dict: + if node_topid == subtrees_topids[subtree]: + node.name = subtree + return sptree,subtrees_dict,subtrees_topids + +#Summarize concordance and conflict from Phyparts +def get_concord_and_conflict(phyparts_root,subtrees_dict,subtrees_topids): + + with open(phyparts_root + ".concon.tre") as phyparts_trees: + concon_tree = Tree(phyparts_trees.readline()) + conflict_tree = Tree(phyparts_trees.readline()) + + concord_dict = {} + conflict_dict = {} + + + for node in concon_tree.traverse(): + node_topid = node.get_topology_id() + for subtree in subtrees_dict: + if node_topid == subtrees_topids[subtree]: + concord_dict[subtree] = node.support + + for node in conflict_tree.traverse(): + node_topid = node.get_topology_id() + for subtree in subtrees_dict: + if node_topid == subtrees_topids[subtree]: + conflict_dict[subtree] = node.support + return concord_dict, conflict_dict + +#Generate Pie Chart data +def get_pie_chart_data(phyparts_root,total_genes,concord_dict,conflict_dict): + + phyparts_hist = [line for line in open(phyparts_root + ".hist")] + phyparts_pies = {} + phyparts_dict = {} + + for n in phyparts_hist: + n = n.split(",") + tot_genes = float(n.pop(-1)) + node_name = n.pop(0)[4:] + concord = float(n.pop(0)) + concord = concord_dict[node_name] + all_conflict = conflict_dict[node_name] + + if len(n) > 0: + most_conflict = max([float(x) for x in n]) + else: + most_conflict = 0.0 + + adj_concord = (concord/total_genes) * 100 + adj_most_conflict = (most_conflict/total_genes) * 100 + other_conflict = (all_conflict - most_conflict) / total_genes * 100 + the_rest = (total_genes - concord - all_conflict) / total_genes * 100 + + pie_list = [adj_concord,adj_most_conflict,other_conflict,the_rest] + + phyparts_pies[node_name] = pie_list + + phyparts_dict[node_name] = [int(round(concord,0)),int(round(tot_genes-concord,0))] + + return phyparts_dict, phyparts_pies + + +def node_text_layout(mynode): + F = faces.TextFace(mynode.name,fsize=20) + faces.add_face_to_node(F,mynode,0,position="branch-right") + +#convert internal phypartspiechart.py data files to csv and export to current directory (for use as ggtree tree data in R) +def pie_data_to_csv(phyparts_dict, phyparts_pies): + phyparts_dist_bin = {} + phyparts_pies_bin = {} + dist_replaced = {} + pies_replaced = {} + + phyparts_dist_bin = json.dumps(phyparts_dist) + phyparts_pies_bin = json.dumps(phyparts_pies) + + + dist_replaced = re.sub(r'{',r'node,concord,genes-concord\n',phyparts_dist_bin) + dist_replaced = re.sub(r'"(\d*)":\s\[(\d*),\s(\d*)\],\s', r'\1,\2,\3\n', dist_replaced) + dist_replaced = re.sub(r'"(\d*)":\s\[(\d*),\s(\d*)\]}', r'\1,\2,\3', dist_replaced) + + pies_replaced = re.sub(r'{',r'node,adj_concord,adj_most_conflict,other_conflict,the_rest\n',phyparts_pies_bin) + pies_replaced = re.sub(r'"(\d*)":\s\[(\d*.\d*),\s(\d*.\d*),\s(\d*.\d*),\s(\d*.\d*)\],\s', r'\1,\2,\3,\4,\5\n', pies_replaced) + pies_replaced = re.sub(r'"(\d*)":\s\[(\d*.\d*),\s(\d*.\d*),\s(\d*.\d*),\s(\d*.\d*)\]}', r'\1,\2,\3,\4,\5', pies_replaced) + + with open('phyparts_dist.csv','w') as file: + for line in dist_replaced: + file.write(line) + with open('phyparts_pies.csv','w') as file: + for line in pies_replaced: + file.write(line) + + +parser = argparse.ArgumentParser(description=helptext,formatter_class=argparse.RawTextHelpFormatter) +parser.add_argument('species_tree',help="Newick formatted species tree topology.") +parser.add_argument('phyparts_root',help="File root name used for Phyparts.") +parser.add_argument('num_genes',type=int,default=0,help="Number of total gene trees. Used to properly scale pie charts.") +parser.add_argument('--taxon_subst',help="Comma-delimted file to translate tip names.") +parser.add_argument("--svg_name",help="File name for SVG generated by script",default="pies.svg") +parser.add_argument("--show_nodes",help="Also show tree with nodes labeled same as PhyParts",action="store_true",default=False) +parser.add_argument("--colors",help="Four colors of the pie chart: concordance (blue) top conflict (green), other conflict (red), no signal (gray)",nargs="+",default=["blue","green","red","dark gray"]) +parser.add_argument("--no_ladderize",help="Do not ladderize the input species tree.",action="store_true",default=False) +parser.add_argument("--to_csv",help="Output data files to csv for import into ggtree in R",action="store_true",default=False) + +args = parser.parse_args() +if args.no_ladderize: + ladderize=False +else: + ladderize=True +plot_tree,subtrees_dict,subtrees_topids = get_phyparts_nodes(args.species_tree, args.phyparts_root) +#print(subtrees_dict) +concord_dict, conflict_dict = get_concord_and_conflict(args.phyparts_root,subtrees_dict,subtrees_topids) +phyparts_dist, phyparts_pies = get_pie_chart_data(args.phyparts_root,args.num_genes,concord_dict,conflict_dict) + +if args.taxon_subst: + taxon_subst = {line.split(",")[0]:line.rstrip().split(",")[1] for line in open(args.taxon_subst,'U')} + for leaf in plot_tree.get_leaves(): + try: + leaf.name = taxon_subst[leaf.name] + except KeyError: + print(leaf.name) + continue +def phyparts_pie_layout(mynode): + if mynode.name in phyparts_pies: + pie= faces.PieChartFace(phyparts_pies[mynode.name], + #colors=COLOR_SCHEMES["set1"], + colors = args.colors, + width=50, height=50) + pie.border.width = None + pie.opacity = 1 + faces.add_face_to_node(pie,mynode, 0, position="branch-right") + + concord_text = faces.TextFace(str(int(concord_dict[mynode.name]))+' ',fsize=20) + conflict_text = faces.TextFace(str(int(conflict_dict[mynode.name]))+' ',fsize=20) + + faces.add_face_to_node(concord_text,mynode,0,position = "branch-top") + faces.add_face_to_node(conflict_text,mynode,0,position="branch-bottom") + + + else: + F = faces.TextFace(mynode.name,fsize=20) + faces.add_face_to_node(F,mynode,0,position="aligned") + +#Plot Pie Chart +ts = TreeStyle() +ts.show_leaf_name = False + +ts.layout_fn = phyparts_pie_layout +nstyle = NodeStyle() +nstyle["size"] = 0 +for n in plot_tree.traverse(): + n.set_style(nstyle) + n.img_style["vt_line_width"] = 0 + +ts.draw_guiding_lines = True +ts.guiding_lines_color = "black" +ts.guiding_lines_type = 0 +ts.scale = 30 +ts.branch_vertical_margin = 10 +plot_tree.convert_to_ultrametric() +if args.to_csv: + pie_data_to_csv(phyparts_dist, phyparts_pies) + +if ladderize: + plot_tree.ladderize(direction=1) +my_svg = plot_tree.render(args.svg_name,tree_style=ts,w=595,dpi=300) + +if args.show_nodes: + node_style = TreeStyle() + node_style.show_leaf_name=False + node_style.layout_fn = node_text_layout + plot_tree.render("tree_nodes.pdf",tree_style=node_style) + +vdisplay.stop() + \ No newline at end of file diff --git a/99.scripts/miscs/plastid/consensus_fasta_from_bam.py b/99.scripts/miscs/plastid/consensus_fasta_from_bam.py new file mode 100644 index 0000000..c19d763 --- /dev/null +++ b/99.scripts/miscs/plastid/consensus_fasta_from_bam.py @@ -0,0 +1,2 @@ +#!/usr/bin/env python3 +"samtools consensus -r NC_058887.1:11845-11988 --show-del yes -A -H 0.3 -m simple --show-ins yes -f fasta EP.sorted.bam" \ No newline at end of file diff --git a/99.scripts/miscs/rename_trinity_fasta.py b/99.scripts/miscs/rename_trinity_fasta.py new file mode 100755 index 0000000..294cab4 --- /dev/null +++ b/99.scripts/miscs/rename_trinity_fasta.py @@ -0,0 +1,71 @@ +#!/usr/bin/env python3 +""" +Trinity FASTA Sequence Renaming Script +Function: Rename sequences in FASTA file to format: [prefix@sequence_number] +""" + +import sys +import os + +def rename_fasta_sequences(input_file, prefix, output_file=None): + """ + Rename sequence headers in a FASTA file + + Parameters: + input_file: Input FASTA filename + prefix: Prefix for sequence names + output_file: Output filename (optional, defaults to input_file_renamed.fasta) + """ + + # Set output filename + if output_file is None: + file_base, file_ext = os.path.splitext(input_file) + output_file = f"{file_base}_renamed{file_ext}" + + # Counter for sequences + seq_count = 0 + + try: + with open(input_file, 'r') as fin, open(output_file, 'w') as fout: + for line in fin: + if line.startswith('>'): + # Sequence header line: rename it + seq_count += 1 + new_name = f">{prefix}@mrna_{seq_count}\n" + fout.write(new_name) + else: + # Sequence data line: write as-is + fout.write(line) + + print(f"Successfully renamed {seq_count} sequences") + print(f"Input file: {input_file}") + print(f"Output file: {output_file}") + print(f"Naming format: {prefix}@mrna_number") + + except FileNotFoundError: + print(f"Error: Input file '{input_file}' not found") + sys.exit(1) + except Exception as e: + print(f"Error processing file: {e}") + sys.exit(1) + +def main(): + """Main function""" + if len(sys.argv) < 3: + print("Usage: python script.py [output_file]") + print("Example: python script.py sequences.fasta Gene new_sequences.fasta") + sys.exit(1) + + input_file = sys.argv[1] + prefix = sys.argv[2] + output_file = sys.argv[3] if len(sys.argv) > 3 else None + + # Verify input file exists + if not os.path.isfile(input_file): + print(f"Error: File '{input_file}' does not exist") + sys.exit(1) + + rename_fasta_sequences(input_file, prefix, output_file) + +if __name__ == "__main__": + main() diff --git a/99.scripts/phyparts b/99.scripts/phyparts new file mode 160000 index 0000000..37067bc --- /dev/null +++ b/99.scripts/phyparts @@ -0,0 +1 @@ +Subproject commit 37067bc3bc14cb34e1e722cd1ac4e12f77613c8a diff --git a/99.scripts/ticr/bucky.pl b/99.scripts/ticr/bucky.pl new file mode 100755 index 0000000..3a2f3f4 --- /dev/null +++ b/99.scripts/ticr/bucky.pl @@ -0,0 +1,1503 @@ +#!/usr/bin/perl +use strict; +use warnings; +use POSIX; +use IO::Select; +use IO::Socket; +use Digest::MD5; +use Getopt::Long; +use Cwd qw(abs_path); +use Fcntl qw(:flock SEEK_END); +use POSIX qw(ceil :sys_wait_h); +use File::Path qw(remove_tree); +use Time::HiRes qw(time usleep); + +# Get OS name +my $os_name = $^O; + +# Turn on autoflush +$|++; + +# Max number of forks to use +my $max_forks = get_free_cpus(); + +# Server port +my $port = 10003; + +# Stores executing machine hostnames +my @machines; +my %machines; + +# Path to text file containing computers to run on +my $machine_file_path; + +# MrBayes block which will be used for each run +my $mb_block; + +# Where this script is located +my $script_path = abs_path($0); + +# Directory script was called from +my $init_dir = abs_path("."); + +# Where the script was called from +my $initial_directory = $ENV{PWD}; + +# General script settings +my $no_forks; + +# Allow for reusing info from an old run +my $input_is_dir = 0; + +# Allow user to specify mbsum output as input for the script +my $input_is_mbsum = 0; + +# How the script was called +my $invocation = "perl bucky.pl @ARGV"; + +# Name of output directory +my $project_name = "bucky-".int(time()); +#my $project_name = "bucky-dir"; + +# BUCKy settings +my $alpha = 1; +my $ngen = 1000000; + +my @unlink; + +# Read commandline settings +GetOptions( + "no-forks" => \$no_forks, + "machine-file=s" => \$machine_file_path, + "alpha|a=s" => \$alpha, + "ngen|n=i" => \$ngen, + "port=i" => \$port, + "no-mbsum|s" => \$input_is_mbsum, + "n-threads|T=i" => \$max_forks, + "out-dir|o=s" => \$project_name, + "server-ip=s" => \&client, # for internal usage only + "help|h" => sub { print &help; exit(0); }, + "usage" => sub { print &usage; exit(0); }, +); + + +# Get paths to required executables +my $bucky = check_path_for_exec("bucky"); +my $mbsum = check_path_for_exec("mbsum") if (!$input_is_mbsum); + +# Check that BUCKy version >= 1.4.4 +check_bucky_version($bucky); + +my $archive = shift(@ARGV); + +# Some error checking +die "You must specify an archive file.\n\n", &usage if (!defined($archive)); +die "Could not locate '$archive', perhaps you made a typo.\n" if (!-e $archive); +die "Could not locate '$machine_file_path'.\n" if (defined($machine_file_path) && !-e $machine_file_path); +die "Invalid alpha for BUCKy specified, input must be a float or 'infinity'.\n" if ($alpha !~ /(^inf(inity)?)|(^\d+(\.\d+)?$)/i); + +# Input is a previous run directory, reuse information +$input_is_dir++ if (-d $archive); + +# Determine which machines we will run the analyses on +if (defined($machine_file_path)) { + + # Get list of machines + print "Fetching machine names listed in '$machine_file_path'...\n"; + open(my $machine_file, '<', $machine_file_path); + chomp(@machines = <$machine_file>); + close($machine_file); + + # Check that we can connect to specified machines + foreach my $index (0 .. $#machines) { + my $machine = $machines[$index]; + print " Testing connection to: $machine...\n"; + + # Attempt to ssh onto machine with a five second timeout + my $ssh_test = `timeout 5 ssh -v $machine exit 2>&1`; + + # Look for machine's IP in test connection + my $machine_ip; + if ($ssh_test =~ /Connecting to \S+ \[(\S+)\] port \d+\./s) { + $machine_ip = $1; + } + + # Could connect but passwordless login not enabled + if ($ssh_test =~ /Are you sure you want to continue connecting \(yes\/no\)/s) { + print " Connection to $machine failed, removing from list of useable machines (passwordless login not enabled).\n"; + splice(@machines, $index, 1); + } + # Successful connection + elsif (defined($machine_ip)) { + print " Connection to $machine [$machine_ip] successful.\n"; + $machines{$machine} = $machine_ip; + } + # Unsuccessful connection + else { + print " Connection to $machine failed, removing from list of useable machines.\n"; + splice(@machines, $index, 1); + } + } +} + +print "\nScript was called as follows:\n$invocation\n\n"; + +my $archive_root; +my $archive_root_no_ext; +if (!$input_is_dir) { + + # Clean run with no prior output + + # Extract name information from input file + + if ($input_is_mbsum) { + ($archive_root = $archive) =~ s/.*\/(.*)/$1/; + ($archive_root_no_ext = $archive) =~ s/(.*\/)?(.*)(\.tar(\.gz)?$)|(\.tgz$)/$2/; + } + else { + ($archive_root = $archive) =~ s/.*\/(.*)/$1/; + ($archive_root_no_ext = $archive) =~ s/(.*\/)?(.*)(\.mb\.tar(\.gz)?$)|(\.mb\.tgz$)/$2/; + } + die "Could not determine archive root name, did you specify the proper input?\n" if (!defined($archive_root_no_ext)); + + # Initialize working directory + # Remove conditional eventually + mkdir($project_name) || die "Could not create '$project_name'$!.\n" if (!-e $project_name); + + my $archive_abs_path = abs_path($archive); + # Remove conditional eventually + run_cmd("ln -s $archive_abs_path $project_name/$archive_root") if (! -e "$project_name/$archive_root"); +} +else { + + # Prior output available, set relevant variables + + $project_name = $archive; + my @contents = glob("$project_name/*"); + + # Determine the archive name by looking for a symlink + my $found_name = 0; + foreach my $file (@contents) { + if (-l $file) { + $file =~ s/\Q$project_name\E\///; + $archive = $file; + $found_name = 1; + } + } + die "Could not locate archive in '$project_name'.\n" if (!$found_name); + + # Extract name information from input file +# ($archive_root = $archive) =~ s/.*\/(.*)/$1/; +# ($archive_root_no_ext = $archive) =~ s/(.*\/)?(.*)(\.mb\.tar(\.gz)?$)|(\.mb\.tgz$)/$2/; + if ($input_is_mbsum) { + ($archive_root = $archive) =~ s/.*\/(.*)/$1/; + ($archive_root_no_ext = $archive) =~ s/(.*\/)?(.*)(\.tar(\.gz)?$)|(\.tgz$)/$2/; + } + else { + ($archive_root = $archive) =~ s/.*\/(.*)/$1/; + ($archive_root_no_ext = $archive) =~ s/(.*\/)?(.*)(\.mb\.tar(\.gz)?$)|(\.mb\.tgz$)/$2/; + } + die "Could not determine archive root name, is your input file properly named?\n" if (!defined($archive_root_no_ext)); +} + +# The name of the output archive +my $mbsum_archive = "$archive_root_no_ext.mbsum.tar.gz"; +my $bucky_archive = "$archive_root_no_ext.BUCKy.tar"; +my $quartet_output = "$archive_root_no_ext.CFs.csv"; + +$mbsum_archive = $archive if ($input_is_mbsum); + +chdir($project_name); + +# Change how Ctrl+C is interpreted to allow for clean up +$SIG{'INT'} = 'INT_handler'; + +# Define and initialize directories +my $mb_out_dir = "mb-out/"; +my $mb_sum_dir = "mb-sum/"; + +mkdir($mb_out_dir) or die "Could not create '$mb_out_dir': $!.\n" if (!-e $mb_out_dir); +mkdir($mb_sum_dir) or die "Could not create '$mb_sum_dir': $!.\n" if (!-e $mb_sum_dir); + +# Check if completed genes from a previous run exist +my %complete_quartets; +if (-e $bucky_archive && -e $quartet_output) { + print "Archive containing completed quartets found for this dataset found in '$bucky_archive'.\n"; + print "Completed quartets within in this archive will be removed from the job queue.\n\n"; + + # Because it takes longer to append to a tarball than append to a text file, the tarball most + # likely has fewer quartet entries in it, we must therefore account for this ensuring that + # the tarball and csv have the same quartet entries + + # See which quartets in the tarball are complete + chomp(my @complete_quartets_tarball = `tar tf '$init_dir/$project_name/$bucky_archive'`); + + # Add quartets to a hash for easier lookup + my %complete_quartets_tarball; + foreach my $complete_quartet (@complete_quartets_tarball) { + $complete_quartets_tarball{$complete_quartet}++; + } + + # Load csv into memory + #open(my $quartet_output_file, "<", $quartet_output); + open(my $quartet_output_file, "<", "$init_dir/$project_name/$quartet_output"); + (my @quartet_info = <$quartet_output_file>); + close($quartet_output_file); + + # Remove header line + my $header = shift(@quartet_info); + + # Rewrite csv to include only quartets also contained in tarball + #open($quartet_output_file, ">", $quartet_output); + open($quartet_output_file, ">", "$init_dir/$project_name/$quartet_output"); + print {$quartet_output_file} $header; + foreach my $quartet (@quartet_info) { + my ($taxon1, $taxon2, $taxon3, $taxon4) = split(",", $quartet); + my $quartet_name = "$taxon1--$taxon2--$taxon3--$taxon4"; + my $quartet_name_tarball = $quartet_name.".tar.gz"; + + if (exists($complete_quartets_tarball{$quartet_name_tarball})) { + $complete_quartets{$quartet_name}++; + print {$quartet_output_file} $quartet; + } + } + close($quartet_output_file); +} + +my @taxa; +my @genes; +if ($input_is_mbsum) { + + # Unarchive input genes + chomp(@genes = `tar xvf '$init_dir/$project_name/$archive' -C $mb_sum_dir 2>&1`); + @genes = map { s/x //; $_ } @genes if ($os_name eq "darwin"); + + die "No genes found in '$archive'.\n" if (!@genes); + + # Move into MrBayes output directory + chdir($mb_sum_dir); + + # Remove subdirectories that may have been a part of the tarball + my @dirs; + my @gene_roots; + foreach my $gene (@genes) { + (my $gene_root = $gene) =~ s/.*\///; + next if ($gene eq $gene_root); + + # Move mbsum files so they are no longer in subdirectories + if (!-d $gene) { + system("mv '$gene' '$gene_root'"); + push(@gene_roots, $gene_root); + } + else { + push(@dirs, $gene); + #splice(@genes, $index, 1); + } + } + @genes = @gene_roots if (@gene_roots); + + # Remove any subdirectories that may have been in the input + foreach my $dir (@dirs) { + remove_tree($dir); + } + + # Parse taxa present in each gene, determine which are shared across all genes + my %taxa; + foreach my $gene (@genes) { + my @taxa = @{parse_mbsum_taxa($gene)}; + + # Count taxa present + foreach my $taxon (@taxa) { + $taxa{$taxon}++; + } + } + + # Add taxa present in all genes to analysis + foreach my $taxon (keys %taxa) { + if ($taxa{$taxon} == scalar(@genes)) { + push(@taxa, $taxon); + } + } + + # Archive genes + system("tar", "czf", $mbsum_archive, @genes); +} +else { + + # Unarchive input genes + chomp(@genes = `tar xvf '$init_dir/$archive' -C '$mb_out_dir' 2>&1`); + @genes = map { s/x //; $_ } @genes if ($os_name eq "darwin"); + + # Move into MrBayes output directory + chdir($mb_out_dir); + + # Check that each gene has a log file + my %taxa; + foreach my $gene (@genes) { + + # Unzip a single gene + chomp(my @mb_files = `tar xvf '$gene' 2>&1`); + @mb_files = map { s/x //; $_ } @mb_files if ($os_name eq "darwin"); + + # Locate the log file output by MrBayes + my $log_file_name; + foreach my $file (@mb_files) { + if ($file =~ /\.log$/) { + $log_file_name = $file; + last; + } + } + die "Could not locate log file for '$gene'.\n" if (!defined($log_file_name)); + + # Parse log file for run information + my $mb_log = parse_mb_log($log_file_name); + + # Check for taxa present + my @taxa = @{$mb_log->{TAXA}}; + foreach my $taxon (@taxa) { + $taxa{$taxon}++; + } + + # Clean up + unlink(@mb_files); + } + + # Add taxa present in all genes to analysis + foreach my $taxon (keys %taxa) { + if ($taxa{$taxon} == scalar(@genes)) { + push(@taxa, $taxon); + } + } +} + +# Create list of possible quartets +my @quartets = combine(\@taxa, 4); + +my $original_size = scalar(@quartets); + +# Remove completed quartets +if (%complete_quartets) { + foreach my $index (reverse(0 .. $#quartets)) { + my $quartet = $quartets[$index]; + $quartet = join("--", @{$quartet}); + if (exists($complete_quartets{$quartet})) { + splice(@quartets, $index, 1); + } + } +} + +print "Found ".scalar(@taxa)." taxa shared across all genes in this archive, ".scalar(@quartets). + " of $original_size possible quartets will be run using output from ".scalar(@genes)." total genes.\n"; + +# Go back to working directory +chdir(".."); + +# Determine whether or not we need to run mbsum on the specified input +my $should_summarize = 1; +if (-e $mbsum_archive && $input_is_dir && !$input_is_mbsum) { + chomp(my @sums = `tar tf '$mbsum_archive'`) || die "Something appears to be wrong with '$mbsum_archive'.\n"; + + # Check that each gene has actually been summarized, if not redo the summaries + if (scalar(@sums) != scalar(@genes)) { + unlink($mbsum_archive); + } + else { + $should_summarize = 0; + } +} + +$should_summarize = 0 if ($input_is_mbsum); + +# Summarize MrBayes output if needed +if ($should_summarize) { + # Run mbsum on each gene + print "Summarizing MrBayes output for ".scalar(@genes)." genes.\n"; + + my @pids; + foreach my $gene (@genes) { + + # Wait until a CPU is available + until(okay_to_run(\@pids)) {}; + + my $pid; + until (defined($pid)) { $pid = fork(); usleep(30000); } + + # The child fork + if ($pid == 0) { + #setpgrp(); + run_mbsum($gene); + exit(0); + } + else { + push(@pids, $pid); + } + } + + # Wait for all summaries to finish + foreach my $pid (@pids) { + waitpid($pid, 0); + } + undef(@pids); + + # Remove directory storing mb output + remove_tree($mb_out_dir); + + # Archive and zip mb summaries + chdir($mb_sum_dir); + #system("tar", "czf", $mbsum_archive, glob("$archive_root_no_ext*.sum")); + system("tar", "czf", $mbsum_archive, glob("*.sum")); + system("cp", $mbsum_archive, ".."); + chdir(".."); +} + +die "\nAll quartets have already been completed.\n\n" if (!@quartets); + +# Returns the external IP address of this computer +chomp(my $server_ip = `dig +short myip.opendns.com \@resolver1.opendns.com 2>&1`); +if ($server_ip !~ /(?:[0-9]{1,3}\.){3}[0-9]{1,3}/) { + print "Could not determine external IP address, only local clients will be created.\n"; + $server_ip = "127.0.0.1"; +} + +# Initialize a server +my $sock = IO::Socket::INET->new( + LocalPort => $port, + Blocking => 0, + Reuse => 1, + Listen => SOMAXCONN, + Proto => 'tcp') +or die "Could not create server socket: $!.\n"; +$sock->autoflush(1); + +print "Job server successfully created.\n"; + +# Should probably do this earlier +# Determine server hostname and add to machines if none were specified by the user +chomp(my $server_hostname = `hostname`); +if (scalar(@machines) == 0) { + push(@machines, $server_hostname); + $machines{$server_hostname} = "127.0.0.1"; +} +elsif (scalar(@machines) == 1) { + # Check if the user input only the local machine in the config + if ($machines{$machines[0]} eq $server_ip) { + $machines{$machines[0]} = "127.0.0.1"; + } +} + +my @pids; +foreach my $machine (@machines) { + + # Fork and create a client on the given machine + my $pid = fork(); + if ($pid == 0) { + close(STDIN); + close(STDOUT); + close(STDERR); + + (my $script_name = $script_path) =~ s/.*\///; + + # Send over/copy files depending on where analyses will be run + if ($machines{$machine} ne "127.0.0.1" && $machines{$machine} ne $server_ip) { + + # Send MrBayes summaries to remote machines + if ($input_is_mbsum) { + system("scp", "-q", "$mb_sum_dir/$mbsum_archive", $machine.":/tmp"); + } + else { + system("scp", "-q", $mbsum_archive, $machine.":/tmp"); + } + + # Send this script to the machine + system("scp", "-q", $script_path, $machine.":/tmp"); + + # Send BUCKy executable to the machine + system("scp", "-q", $bucky, $machine.":/tmp"); + + # Execute this perl script in client mode on the given machine + # -tt forces pseudo-terminal allocation and lets us stop remote processes + exec("ssh", "-tt", $machine, "perl", "/tmp/$script_name", $mbsum_archive, "--server-ip=$server_ip:$port"); + } + else { + # Send this script to the machine + system("cp", $script_path, "/tmp"); + + # Send BUCKy executable to the machine + system("cp", $bucky, "/tmp"); + + # Execute this perl script in client mode + exec("perl", "/tmp/$script_name", "$init_dir/$project_name/$mb_sum_dir/$mbsum_archive", "--server-ip=127.0.0.1:$port"); + } + + exit(0); + } + else { + push(@pids, $pid); + } +} + +# Move into mbsum directory +chdir($mb_sum_dir); + +my $select = IO::Select->new($sock); + +# Stores which job is next in queue +my $job_number = 0; + +# Number of open connections to a client +my $total_connections; + +# Number of complete jobs (necessary?) +my $complete_count = 0; + +# Number of connections server has closed +my $closed_connections = 0; + +# Minimum number of connections server should expect +my $starting_connections = scalar(@machines); + +my $time = time(); +my $num_digits = get_num_digits({'NUMBER' => scalar(@quartets)}); + +# Begin the server's job distribution +my %complete_queue; +my $complete_queue_max_size = 100; +while ((!defined($total_connections) || $closed_connections != $total_connections) || $total_connections < $starting_connections) { + # Contains handles to clients which have sent information to the server + my @clients = $select->can_read(0); + + # Free up CPU by sleeping for 10 ms + usleep(10000); + + # Reap any children that we can + foreach my $pid (@pids) { + waitpid($pid, WNOHANG); + } + + # Handle each ready client individually + CLIENT: foreach my $client (@clients) { + + if (scalar(keys %complete_queue) > $complete_queue_max_size) { + my $cwd = abs_path("."); + chdir($mb_sum_dir); + dump_quartets(\%complete_queue); + chdir($cwd); + undef(%complete_queue); + } + + # Client requesting new connection + if ($client == $sock) { + $total_connections++; + $select->add($sock->accept()); + } + else { + + # Get client's message + my $response = <$client>; + next if (not defined($response)); # a response should never actually be undefined + + # Client wants to send us a file + if ($response =~ /SEND_FILE: (.*)/) { + my $file_name = $1; + receive_file({'FILE_PATH' => $file_name, 'FILE_HANDLE' => $client}); + } + + # Client has finished a job + if ($response =~ /DONE '(.*)' '(.*)' \|\|/) { + $complete_count++; + printf(" Analyses complete: %".$num_digits."d/%d.\r", $complete_count, scalar(@quartets)); + + my $completed_quartet = $1; + my $quartet_statistics = $2; + + #push(@unlink, glob("$completed_quartet*")); + push(@unlink, glob(abs_path(".")."/$completed_quartet*")); + $complete_queue{$completed_quartet} = $quartet_statistics; + } + + # Client wants a new job + if ($response =~ /NEW: (.*)/) { + my $client_ip = $1; + + # Check if jobs remain in the queue + if ($job_number < scalar(@quartets)) { + printf("\n Analyses complete: %".$num_digits."d/%d.\r", 0, scalar(@quartets)) if ($job_number == 0); + + my $quartet = join("--", @{$quartets[$job_number]}); + + # Tell local clients to move into mbsum directory + if ($client_ip eq $server_ip) { + print {$client} "CHDIR: ".abs_path(".")."\n"; + } + + # Invocation changes if we want to use a prior of infinity + if ($alpha =~ /(^inf(inity)?)/i) { + print {$client} "NEW: '$quartet' '--use-independence-prior -n $ngen'\n"; + } + else { + print {$client} "NEW: '$quartet' '-a $alpha -n $ngen'\n"; + } + $job_number++; + } + else { + # Client has asked for a job, but there are none remaining + print {$client} "HANGUP\n"; + $select->remove($client); + $client->close(); + $closed_connections++; + next CLIENT; + } + } + } + } +} + +# Dump remaining quartets +dump_quartets(\%complete_queue); + +# Wait until all children have completed +foreach my $pid (@pids) { + waitpid($pid, 0); +} + +print "\n All connections closed.\n"; +print "Total execution time: ", sec2human(time() - $time), ".\n\n"; + +rmdir("$initial_directory/$project_name/$mb_sum_dir"); + +sub client { + my ($opt_name, $address) = @_; + + my ($server_ip, $port) = split(":", $address); + + chdir("/tmp"); + my $bucky = "/tmp/bucky"; + + my $pgrp = $$; + setpgrp(); + + # Determine this host's IP + chomp(my $ip = `dig +short myip.opendns.com \@resolver1.opendns.com`); + + # Set IP to localhost if we don't have internet + if ($ip !~ /(?:[0-9]{1,3}\.){3}[0-9]{1,3}/) { + $ip = "127.0.0.1"; + } + + # Determine file name of mbsum archive the client should use + my @ARGV = split(/\s+/, $invocation); + shift(@ARGV); shift(@ARGV); # remove "perl" and "bucky.pl" + my $mbsum_archive = shift(@ARGV); + + # Extract files from mbsum archive + my @sums; + if ($mbsum_archive =~ /\//) { + chomp(@sums = `tar tf '$mbsum_archive'`); + } + else { + chomp(@sums = `tar xvf '$mbsum_archive' 2>&1`); + @sums = map { s/x //; $_ } @sums if ($os_name eq "darwin"); + } + + # Spawn more clients + my @pids; + # A slightly modification in order to limit forks to 10 + # my $total_forks = get_free_cpus(); + my $total_forks = 10; + if ($total_forks > 1) { + foreach my $fork (1 .. $total_forks - 1) { + my $pid = fork(); + if ($pid == 0) { + last; + } + else { + push(@pids, $pid); + } + } + } + + # The name of the quartet we are working on + my $quartet; + + # Stores names of unneeded files + my @unlink; + + # Change signal handling so killing the server kills these processes and cleans up + $SIG{HUP} = sub { unlink($0, $bucky); unlink(@sums); unlink($mbsum_archive); kill -15, $$; }; + $SIG{TERM} = sub { unlink(glob("$quartet*")) if (defined($quartet)); exit(0); }; + + # Connect to the server + my $sock = new IO::Socket::INET( + PeerAddr => $server_ip.":".$port, + Proto => 'tcp') + or exit(0); + $sock->autoflush(1); + + print {$sock} "NEW: $ip\n"; + while (chomp(my $response = <$sock>)) { + + if ($response =~ /CHDIR: (.*)/) { + chdir($1); + } + elsif ($response =~ /NEW: '(.*)' '(.*)'/) { + $quartet = $1; + my $bucky_settings = $2; + + # If client is local this needs to be defined now + #chomp(@sums = `tar tf $mbsum_archive`) if (!@sums); + + # Create prune tree file contents required for BUCKy + my $count = 0; + my $prune_tree_output = "translate\n"; + foreach my $member (split("--", $quartet)) { + $count++; + $prune_tree_output .= " $count $member"; + if ($count == 4) { + $prune_tree_output .= ";\n"; + } + else { + $prune_tree_output .= ",\n"; + } + } + + # Write prune tree file + my $prune_file_path = "$quartet-prune.txt"; + + push(@unlink, $prune_file_path); + + open(my $prune_file, ">", $prune_file_path); + print {$prune_file} $prune_tree_output; + close($prune_file); + + push(@unlink, "$quartet.input", "$quartet.out", "$quartet.cluster", "$quartet.concordance", "$quartet.gene"); + + # Run BUCKy on specified quartet + system($bucky, split(" ", $bucky_settings), "-cf", 0, "-o", $quartet, "-p", $prune_file_path, @sums); + unlink($prune_file_path); + + # Zip and tarball the results + my @results = glob($quartet."*"); + my $quartet_archive_name = "$quartet.tar.gz"; + + # Open concordance file and parse out the three possible resolutions + my $num_genes = get_used_genes("$quartet.out"); + my $split_info = parse_concordance_output("$quartet.concordance", $num_genes); + + # Archive and compress results + system("tar", "czf", $quartet_archive_name, @results); + + # Send the results back to the server if this is a remote client + if ($server_ip ne "127.0.0.1" && $server_ip ne $ip) { + send_file({'FILE_PATH' => $quartet_archive_name, 'FILE_HANDLE' => $sock}); + unlink($quartet_archive_name); + } + + unlink(@unlink); + undef(@unlink); + + print {$sock} "DONE '$quartet_archive_name' '$split_info' || NEW: $ip\n"; + } + elsif ($response eq "HANGUP") { + last; + } + } + + # Have initial client wait for all others to finish and clean up + if ($$ == $pgrp) { + foreach my $pid (@pids) { + waitpid($pid, 0); + } + unlink($0, $bucky); + unlink(@sums, $mbsum_archive); + #if ($server_ip ne $ip) { + # unlink(@sums, $mbsum_archive); + #} + #else { + # unlink(@sums); + #} + } + + exit(0); +} + +sub dump_quartets { + my $quartets = shift; + + my $pid; + until (defined($pid)) { $pid = fork(); usleep(30000); } + if ($pid == 0) { + + my @completed_quartets = keys %{$quartets}; + my @quartet_statistics = values %{$quartets}; + + $SIG{TERM} = sub { close(STDIN); close(STDOUT); close(STDERR); unlink(@unlink); exit(0); }; + + # Check if this is the first to complete, if so we must create CF output file + if (!-e "../$quartet_output") { + open(my $quartet_output_file, ">", "../$quartet_output"); + print {$quartet_output_file} "taxon1,taxon2,taxon3,taxon4,CF12_34,CF12_34_lo,CF12_34_hi,CF13_24,CF13_24_lo,CF13_24_hi,CF14_23,CF14_23_lo,CF14_23_hi,ngenes\n"; + foreach my $quartet (@quartet_statistics) { + print {$quartet_output_file} $quartet,"\n"; + } + close($quartet_output); + } + else { + # Obtain a file lock on archive so another process doesn't simultaneously try to add to it + open(my $quartet_output_file, ">>", "../$quartet_output"); + flock($quartet_output_file, LOCK_EX) || die "Could not lock '$quartet_output': $!.\n"; + seek($quartet_output_file, 0, SEEK_END) || die "Could not seek '$quartet_output': $!.\n"; + + # Add completed gene + #print {$quartet_output_file} @quartet_statistics,"\n"; + foreach my $quartet (@quartet_statistics) { + print {$quartet_output_file} $quartet,"\n"; + } + + # Release lock + flock($quartet_output_file, LOCK_UN) || die "Could not unlock '$quartet_output': $!.\n"; + close($quartet_output_file); + } + + # Check if this is the first to complete, if so we must create BUCKy tarball + # if (!-e "../$bucky_archive") { + # system("tar", "cf", "../$bucky_archive", @completed_quartets); + # unlink(@completed_quartets); + # } + # else { + # my @unlink; + # foreach my $completed_quartet (@completed_quartets) { + # (my $quartet = $completed_quartet) =~ s/\.tar\.gz//; + # push(@unlink, glob("$quartet*")); + # } + # #$SIG{TERM} = sub { close(STDIN); close(STDOUT); close(STDERR); unlink(glob("$quartet*")); exit(0); }; + # #$SIG{TERM} = sub { close(STDIN); close(STDOUT); close(STDERR); unlink(@unlink); exit(0); }; + + # # Obtain a file lock on archive so another process doesn't simultaneously try to add to it + # open(my $bucky_archive_file, "<", "../$bucky_archive"); + # flock($bucky_archive_file, LOCK_EX) || die "Could not lock '$bucky_archive': $!.\n"; + + # # Add completed gene + # system("tar", "rf", "../$bucky_archive", @completed_quartets); + # unlink(@completed_quartets); + + # # Release lock + # flock($bucky_archive_file, LOCK_UN) || die "Could not unlock '$bucky_archive': $!.\n"; + # close($bucky_archive_file); + # } + + unlink(@unlink); + undef(@unlink); + + exit(0); + } + else { + push(@pids, $pid); + } + +# unlink(@unlink); +# undef(@unlink); + + return; +} + +sub get_used_genes { + my $file_name = shift; + + # Number of genes actually used by BUCKy + my $num_genes; + + # Open up the BUCKy output file + open(my $bucky_out, "<", $file_name); + while (my $line = <$bucky_out>) { + if ($line =~ /Read (\d+) genes with a total of/) { + $num_genes = $1; + last; + } + } + close($bucky_out); + + die "Error determining number of genes used by BUCKy ($file_name).\n" if (!defined($num_genes)); + + return $num_genes; +} + +sub parse_concordance_output { + my ($file_name, $ngenes) = @_; + + my @taxa; + my %splits; + + # Open up the specified output file + open(my $concordance_file, "<", $file_name); + + my $split; + my $in_translate; + my $in_all_splits; + while (my $line = <$concordance_file>) { + + # Parse the translate table + if ($in_translate) { + if ($line =~ /\d+ (.*)([,;])/) { + my $taxon = $1; + my $line_end = $2; + push(@taxa, $taxon); + + $in_translate = 0 if ($line_end eq ';'); + } + } + + # Parse the split information + if ($in_all_splits) { + + # Set the split we are parsing information from + if ($line =~ /^(\{\S+\})/) { + my $current_split = $1; + if ($current_split eq "{1,4|2,3}") { + $split = "14|23"; + } + elsif ($current_split eq "{1,3|2,4}") { + $split = "13|24"; + } + elsif ($current_split eq "{1,2|3,4}") { + $split = "12|34"; + } + } + + # Parse mean number of loci for split + if ($line =~ /=\s+(\S+) \(number of loci\)/) { + $splits{$split}->{"CF"} = $1 / $ngenes; + } + + # Parse 95% confidence interval + if ($line =~ /95% CI for CF = \((\d+),(\d+)\)/) { + #$splits{$split}->{"95%_CI"} = "(".($1 / $ngenes).",".($2 / $ngenes).")"; + $splits{$split}->{"95%_CI_LO"} = ($1 / $ngenes); + $splits{$split}->{"95%_CI_HI"} = ($2 / $ngenes); + } + } + + $in_translate++ if ($line =~ /^translate/); + $in_all_splits++ if ($line =~ /^All Splits:/); + } + + # Concat taxa names together + my $return = join(",", @taxa); + $return .= ","; + + # Concat split proportions with their 95% CI to return + if (exists($splits{"12|34"})) { + #$return .= $splits{"12|34"}->{"CF"}.$splits{"12|34"}->{"95%_CI"}."\t"; + $return .= $splits{"12|34"}->{"CF"}.",".$splits{"12|34"}->{"95%_CI_LO"}.",".$splits{"12|34"}->{"95%_CI_HI"}.","; + } + else { + #$return .= "0(0,0)\t"; + $return .= "0,0,0,"; + } + + if (exists($splits{"13|24"})) { + #$return .= $splits{"13|24"}->{"CF"}.$splits{"13|24"}->{"95%_CI"}."\t"; + $return .= $splits{"13|24"}->{"CF"}.",".$splits{"13|24"}->{"95%_CI_LO"}.",".$splits{"13|24"}->{"95%_CI_HI"}.","; + } + else { + #$return .= "0(0,0)\t"; + $return .= "0,0,0,"; + } + + if (exists($splits{"14|23"})) { + #$return .= $splits{"14|23"}->{"CF"}.$splits{"14|23"}->{"95%_CI"}; + $return .= $splits{"14|23"}->{"CF"}.",".$splits{"14|23"}->{"95%_CI_LO"}.",".$splits{"14|23"}->{"95%_CI_HI"}; + } + else { + #$return .= "0(0,0)"; + $return .= "0,0,0"; + } + + # Append number of genes used + $return .= ",$ngenes"; + + return $return; +} + +sub parse_mb_log { + my $log_file_name = shift; + + # Open the specified mb log file and parse useful information from it + + my @taxa; + my $ngen; + my $nruns; + my $burnin; + my $burninfrac; + my $samplefreq; + open(my $log_file, "<", $log_file_name); + while (my $line = <$log_file>) { + if ($line =~ /Taxon\s+\d+ -> (\S+)/) { + push(@taxa, $1); + } + elsif ($line =~ /Setting number of runs to (\d+)/) { + $nruns = $1; + } + elsif ($line =~ /Setting burnin fraction to (\S+)/) { + $burninfrac = $1; + } + elsif ($line =~ /Setting chain burn-in to (\d+)/) { + $burnin = $1; + } + elsif ($line =~ /Setting sample frequency to (\d+)/) { + $samplefreq = $1; + } + elsif ($line =~ /Setting number of generations to (\d+)/) { + $ngen = $1; + } + } + close($log_file); + + if (defined($burnin)) { + #return {'SAMPLEFREQ' => $samplefreq, 'NRUNS' => $nruns, 'BURNIN' => $burnin }; + return {'SAMPLEFREQ' => $samplefreq, 'NRUNS' => $nruns, 'BURNIN' => $burnin, 'TAXA' => \@taxa}; + } + else { + return {'NGEN' => $ngen, 'NRUNS' => $nruns, 'BURNINFRAC' => $burninfrac, + 'SAMPLEFREQ' => $samplefreq, 'TAXA' => \@taxa}; + } +} + +sub parse_mbsum_taxa { + my $mbsum_file_name = shift; + + # Open the specified mbsum file and parse its taxa list + + my @taxa; + my $in_translate_block = 0; + open(my $mbsum_file, "<", $mbsum_file_name); + while (my $line = <$mbsum_file>) { + $in_translate_block++ if ($line =~ /translate/); + + if ($in_translate_block == 1 && $line =~ /\d+\s+([^,;]+)/) { + push(@taxa, $1); + } + } + close($mbsum_file); + + # Check if there were multiple translate blocks in the file which is indicative of an error with the creation of the file + die "Something is amiss with '$mbsum_file_name', multiple translate blocks ($in_translate_block) were detected when there should only be one.\n" if ($in_translate_block > 1); + + # Check that we actually parsed something, otherwise input is improperly formatted/not mbsum output + die "No taxa parsed for file '$mbsum_file_name', does '$archive' actually contain mbsum output?.\n" if (!@taxa); + + return \@taxa; +} + +sub run_mbsum { + my $tarball = shift; + + # Unzip specified tarball + chomp(my @mb_files = `tar xvf '$mb_out_dir$tarball' -C '$mb_out_dir' 2>&1`); + @mb_files = map { s/x //; $_ } @mb_files if ($os_name eq "darwin"); + + # Determine name for this partition's log file + my $log_file_name; + foreach my $file (@mb_files) { + if ($file =~ /\.log$/) { + $log_file_name = $file; + last; + } + } + die "Could not locate log file for '$tarball'.\n" if (!defined($log_file_name)); + + # Parse log file + my $mb = parse_mb_log("$mb_out_dir$log_file_name"); + + (my $gene_name = $tarball) =~ s/\.nex\.tar\.gz//; + + # Determine number of trees mbsum should remove from each file + + my $trim; + if ($mb->{BURNIN}) { + $trim = $mb->{BURNIN} + 1; + } + else { + $trim = ((($mb->{NGEN} / $mb->{SAMPLEFREQ}) * $mb->{NRUNS} * $mb->{BURNINFRAC}) / $mb->{NRUNS}) + 1; + } + + # Summarize gene's tree files + system("$mbsum '$mb_out_dir$gene_name.'*.t -n $trim -o '$mb_sum_dir$gene_name.sum' >/dev/null 2>&1"); + + # Clean up extracted files + chdir($mb_out_dir); + unlink(@mb_files); +} + +sub okay_to_run { + my $pids = shift; + + # Free up a CPU by sleeping for 10 ms + usleep(10000); + + my $current_forks = scalar(@{$pids}); + foreach my $index (reverse(0 .. $current_forks - 1)) { + next if ($index < 0); + + my $pid = @{$pids}[$index]; + my $wait = waitpid($pid, WNOHANG); + + # Successfully reaped child + if ($wait > 0) { + $current_forks--; + #splice(@pids, $index, 1); + splice(@{$pids}, $index, 1); + } + } + + return ($current_forks < $max_forks); +} + +sub hashsum { + my $settings = shift; + + my $file_path = $settings->{'FILE_PATH'}; + + open(my $file, "<", $file_path) or die "Couldn't open file '$file_path': $!.\n"; + my $md5 = Digest::MD5->new; + my $md5sum = $md5->addfile(*$file)->hexdigest; + close($file); + + return $md5sum; +} + +sub send_file { + my $settings = shift; + + my $file_path = $settings->{'FILE_PATH'}; + my $file_handle = $settings->{'FILE_HANDLE'}; + + my $hash = hashsum({'FILE_PATH' => $file_path}); + print {$file_handle} "SEND_FILE: $file_path\n"; + + open(my $file, "<", $file_path) or die "Couldn't open file '$file_path': $!.\n"; + while (<$file>) { + print {$file_handle} $_; + } + close($file); + + print {$file_handle} " END_FILE: $hash\n"; + + # Stall until we know status of file transfer + while (defined(my $response = <$file_handle>)) { + chomp($response); + + last if ($response eq "TRANSFER_SUCCESS"); + die "Unsuccessful file transfer, checksums did not match.\n" if ($response eq "TRANSFER_FAILURE"); + } +} + +sub receive_file { + my $settings = shift; + + my $file_path = $settings->{'FILE_PATH'}; + my $file_handle = $settings->{'FILE_HANDLE'}; + + my $check_hash; + open(my $file, ">", $file_path); + while (<$file_handle>) { + if ($_ =~ /(.*) END_FILE: (\S+)/) { + print {$file} $1; + $check_hash = $2; + last; + } + else { + print {$file} $_; + } + } + close($file); + + # Use md5 hashsum to make sure transfer worked + my $hash = hashsum({'FILE_PATH' => $file_path}); + if ($hash ne $check_hash) { + die "Unsuccessful file transfer, checksums do not match.\n'$hash' - '$check_hash'\n"; # hopefully this never pops up + print {$file_handle} "TRANSFER_FAILURE\n" + } + + else { + print {$file_handle} "TRANSFER_SUCCESS\n"; + } +} + +sub INT_handler { + #dump_quartets(\%complete_queue); + + unlink(@unlink); + + # Kill ssh process(es) spawned by this script + foreach my $pid (@pids) { + #kill(-9, $pid); + #kill(15, $pid); + kill(1, $pid); + } + + # Move into gene directory + chdir("$initial_directory/$project_name"); + + rmdir($mb_out_dir); + + # Try to delete directory once per second for five seconds, if it can't be deleted print an error message + # I've found this method is necessary for analyses performed on AFS drives + my $count = 0; + until (!-e $mb_sum_dir || $count == 5) { + $count++; + + remove_tree($mb_sum_dir, {error => \my $err}); + sleep(1); + } + #logger("Could not clean all files in './$gene_dir/'.") if ($count == 5); + print "Could not clean all files in './$mb_sum_dir/'.\n" if ($count == 5); + + exit(0); +} + +sub clean_up { + my $settings = shift; + + my $remove_dirs = $settings->{'DIRS'}; + my $current_dir = getcwd(); + +# chdir($alignment_root); +# unlink(glob($gene_dir."$alignment_name*")); +# #unlink($server_check_file) if (defined($server_check_file)); +# +# if ($remove_dirs) { +# rmdir($gene_dir); +# } + chdir($current_dir); +} + +sub get_num_digits { + my $settings = shift; + + my $number = $settings->{'NUMBER'}; + + my $digits = 1; + while (floor($number / 10) != 0) { + $number = floor($number / 10); + $digits++; + } + + return $digits; +} + +sub sec2human { + my $secs = shift; + + # Constants + my $secs_in_min = 60; + my $secs_in_hour = 60 * 60; + my $secs_in_day = 24 * 60 * 60; + + $secs = int($secs); + + return "0 seconds" if (!$secs); + + # Calculate units of time + my $days = int($secs / $secs_in_day); + my $hours = ($secs / $secs_in_hour) % 24; + my $mins = ($secs / $secs_in_min) % 60; + $secs = $secs % 60; + + # Format return nicely + my $time; + if ($days) { + $time .= ($days != 1) ? "$days days, " : "$days day, "; + } + if ($hours) { + $time .= ($hours != 1) ? "$hours hours, " : "$hours hour, "; + } + if ($mins) { + $time .= ($mins != 1) ? "$mins minutes, " : "$mins minute, "; + } + if ($secs) { + $time .= ($secs != 1) ? "$secs seconds " : "$secs second "; + } + else { + # Remove comma + chop($time); + } + chop($time); + + return $time; +} + +sub get_free_cpus { + + my $os_name = $^O; + + # Returns a two-member array containing CPU usage observed by top, + # top is run twice as its first output is usually inaccurate + my @percent_free_cpu; + if ($os_name eq "darwin") { + # Mac OS + chomp(@percent_free_cpu = `top -i 1 -l 2 | grep "CPU usage"`); + } + else { + # Linux + chomp(@percent_free_cpu = `top -b -n2 -d0.05 | grep "Cpu(s)"`); + } + + my $percent_free_cpu = pop(@percent_free_cpu); + + if ($os_name eq "darwin") { + # Mac OS + $percent_free_cpu =~ s/.*?(\d+\.\d+)%\s+id.*/$1/; + } + else { + # linux + $percent_free_cpu =~ s/.*?(\d+\.\d)\s*%?ni,\s*(\d+\.\d)\s*%?id.*/$1 + $2/; # also includes %nice as free + $percent_free_cpu = eval($percent_free_cpu); + } + + my $total_cpus; + if ($os_name eq "darwin") { + # Mac OS + $total_cpus = `sysctl -n hw.ncpu`; + } + else { + # linux + $total_cpus = `grep --count 'cpu' /proc/stat` - 1; + } + + my $free_cpus = ceil($total_cpus * $percent_free_cpu / 100); + + if ($free_cpus == 0 || $free_cpus !~ /^\d+$/) { + $free_cpus = 1; # assume that at least one cpu can be used + } + + return $free_cpus; +} + +sub run_cmd { + my $command = shift; + + my $return = system($command); + + if ($return) { + logger("'$command' died with error: '$return'.\n"); + #kill(2, $parent_pid); + exit(0); + } +} + +sub check_path_for_exec { + my $exec = shift; + + my $path = $ENV{PATH}.":."; # include current directory as well + my @path_dirs = split(":", $path); + + my $exec_path; + foreach my $dir (@path_dirs) { + $dir .= "/" if ($dir !~ /\/$/); + $exec_path = abs_path($dir.$exec) if (-e $dir.$exec && -x $dir.$exec && !-d $dir.$exec); + } + + die "Could not find the following executable: '$exec'. This script requires this program in your path.\n" if (!defined($exec_path)); + return $exec_path; +} + +# I grabbed this from StackOverflow so that's why its style is different #DontFixWhatIsntBroken: +# https://stackoverflow.com/questions/10299961/in-perl-how-can-i-generate-all-possible-combinations-of-a-list +sub combine { + my ($list, $n) = @_; + die "Insufficient list members" if ($n > @$list); + + return map [$_], @$list if ($n <= 1); + + my @comb; + + for (my $i = 0; $i+$n <= @$list; ++$i) { + my $val = $list->[$i]; + my @rest = @$list[$i + 1 .. $#$list]; + push(@comb, [$val, @$_]) for combine(\@rest, $n - 1); + } + + return @comb; +} + +sub check_bucky_version { + my $bucky = shift; + + print "\nChecking for BUCKy version >= 1.4.4...\n"; + + # Run BUCKy with --version and extract version info + chomp(my @version_info = grep { /BUCKy version/ } `$bucky --version`); + my $version_info = shift(@version_info); + + die " Could not determine BUCKy version.\n" if (!defined($version_info)); + + # Get the actual version number + my $version; + if ($version_info =~ /BUCKy\s+version\s+([^\s|,]+)/) { + $version = $1; + } + die " Could not determine BUCKy version.\n" if (!defined($version_info)); + + # Version testing + #my @versions = qw/2.1b 1.2.1000 1 0.9.8 2.3 1.4.5 1.4.3 1.4 1.500.2 1.4.4 1.4.4.1/; + #foreach my $version (@versions) { + + print " BUCKy version: $version.\n"; + + # Split version number based on period delimiters + my @version_parts = split(/\./, $version); + + # Future proofing if letters are ever used (we won't ever care about them) + @version_parts = map { s/[a-zA-Z]+//g; $_ } @version_parts; + die " Error determining BUCKy version.\n" if (!@version_parts); + + # Check that version is >= 1.4.4 + if (defined($version_parts[0]) && $version_parts[0] > 1) { + print " BUCKy version check passed.\n"; + return; + } + elsif ((defined($version_parts[0]) && $version_parts[0] == 1) && (defined($version_parts[1]) && $version_parts[1] > 4)) { + print " BUCKy version check passed.\n"; + return; + } + elsif (((defined($version_parts[0]) && $version_parts[0] == 1) && (defined($version_parts[1]) && $version_parts[1] == 4)) + && defined($version_parts[2]) && $version_parts[2] >= 4) { + print " BUCKy version check passed.\n"; + return; + } + else { + die " BUCKy version check failed, update to version >= 1.4.4.\n"; + } + + #print "\n"; + #} +} + +sub usage { + return "Usage: bucky.pl [MRBAYES TARBALL]\n"; +} + +sub help { +print < +EOF +exit(0); +} diff --git a/99.scripts/ticr/mb.pl b/99.scripts/ticr/mb.pl new file mode 100755 index 0000000..70aedbc --- /dev/null +++ b/99.scripts/ticr/mb.pl @@ -0,0 +1,1102 @@ +#!/usr/bin/perl +use strict; +use warnings; +use POSIX; +use IO::Select; +use IO::Socket; +use Digest::MD5; +use Getopt::Long; +use Cwd qw(abs_path); +use Fcntl qw(:flock); +use File::Path qw(remove_tree); +use Time::HiRes qw(time usleep); + +my $os_name = $^O; + +# Turn on autoflush +$|++; + +# Maximum number of threads to use +my $max_forks; + +# Server port +my $port = 10002; + +# Stores executing machine hostnames +my @machines; +my %machines; + +# Path to text file containing computers to run on +my $machine_file_path; + +# MrBayes block which will be used for each run +my $mb_block; + +# Where this script is located +my $script_path = abs_path($0); + +# Directory script was called from +my $init_dir = abs_path("."); + +# Where the script was called from +my $initial_directory = $ENV{PWD}; + +# Allow for reusing info from an old run +my $input_is_dir = 0; + +# How the script was called +my $invocation = "perl mb.pl @ARGV"; + +# Name of output directory +my $project_name = "mb-".int(time()); +#my $project_name = "mb-dir"; + +# Read commandline settings +GetOptions( + #"no-forks" => \$no_forks, + "mb-block|m:s" => \$mb_block, + "machine-file:s" => \$machine_file_path, + "check|c:f" => \&check_nonconvergent, + "remove|r:f" => \&remove_nonconvergent, + "out-dir|o=s" => \$project_name, + "n-threads|T" => \$max_forks, + "port=i" => \$port, + "server-ip:s" => \&client, # for internal usage only + "help|h" => sub { print &help; exit(0); }, + "usage" => sub { print &usage; exit(0); }, +); + + +# Get paths to required executables +my $mb = check_path_for_exec("mb"); + +my $archive = shift(@ARGV); + +# Some error checking +die "You must specify an archive file.\n\n", &usage if (!defined($archive)); +die "Could not locate '$archive', perhaps you made a typo.\n" if (!-e $archive); +die "You specified a MrBayes run archive instead of an MDL gene archive.\n" if ($archive =~ /\.mb\.tar$/); +die "Could not locate '$machine_file_path'.\n" if (defined($machine_file_path) && !-e $machine_file_path); +die "You must specify a file containing a valid MrBayes block which will be appended to each gene.\n\n", &usage if (!defined($mb_block)); +die "Could not locate '$mb_block', perhaps you made a typo.\n\n" if (!-e $mb_block); + +# Input is a previous run directory, reuse information +$input_is_dir++ if (-d $archive); + +# Determine which machines we will run the analyses on +if (defined($machine_file_path)) { + + # Get list of machines + print "Fetching machine names listed in '$machine_file_path'...\n"; + open(my $machine_file, '<', $machine_file_path); + chomp(@machines = <$machine_file>); + close($machine_file); + + # Check that we can connect to specified machines + foreach my $index (0 .. $#machines) { + my $machine = $machines[$index]; + print " Testing connection to: $machine...\n"; + + # Attempt to ssh onto machine with a five second timeout + my $ssh_test = `timeout 5 ssh -v $machine exit 2>&1`; + + # Look for machine's IP in test connection + my $machine_ip; + if ($ssh_test =~ /Connecting to \S+ \[(\S+)\] port \d+\./s) { + $machine_ip = $1; + } + + # Could connect but passwordless login not enabled + if ($ssh_test =~ /Are you sure you want to continue connecting \(yes\/no\)/s) { + print " Connection to $machine failed, removing from list of useable machines (passwordless login not enabled).\n"; + splice(@machines, $index, 1); + } + # Successful connection + elsif (defined($machine_ip)) { + print " Connection to $machine [$machine_ip] successful.\n"; + $machines{$machine} = $machine_ip; + } + # Unsuccessful connection + else { + print " Connection to $machine failed, removing from list of useable machines.\n"; + splice(@machines, $index, 1); + } + } +} + +print "\nScript was called as follows:\n$invocation\n"; + +# Load MrBayes block into memory +open(my $mb_block_file, "<", $mb_block) or die "Could not open '$mb_block': $!.\n"; +my @mb_block = <$mb_block_file>; +close($mb_block_file); + +my $archive_root; +my $archive_root_no_ext; +if (!$input_is_dir) { + + # Clean run with no prior output + + # Extract name information from input file + ($archive_root = $archive) =~ s/.*\/(.*)/$1/; + ($archive_root_no_ext = $archive) =~ s/(.*\/)?(.*)(\.tar\.gz)|(\.tgz)/$2/; + + # Initialize working directory + # Remove conditional eventually + mkdir($project_name) || die "Could not create '$project_name'$!.\n" if (!-e $project_name); + + my $archive_abs_path = abs_path($archive); + # Remove conditional eventually + run_cmd("ln -s $archive_abs_path $project_name/$archive_root") if (! -e "$project_name/$archive_root"); +} +else { + + # Prior output available, set relevant variables + + $project_name = $archive; + my @contents = glob("$project_name/*"); + + # Determine the archive name by looking for a symlink + my $found_name = 0; + foreach my $file (@contents) { + if (-l $file) { + $file =~ s/\Q$project_name\E\///; + #$archive = $file; + $archive = "$project_name/$file"; + $found_name = 1; + } + } + die "Could not locate archive in '$project_name'.\n" if (!$found_name); + + # Extract name information from input file + ($archive_root = $archive) =~ s/.*\/(.*)/$1/; + ($archive_root_no_ext = $archive) =~ s/(.*\/)?(.*)(\.tar\.gz)|(\.tgz)/$2/; +} + +# The name of the output archive +my $mb_archive = "$archive_root_no_ext.mb.tar"; + +chdir($project_name); + +# Change how Ctrl+C is interpreted to allow for clean up +$SIG{'INT'} = 'INT_handler'; + +# Define and initialize directories +my $gene_dir = "genes/"; +mkdir($gene_dir) or die "Could not create '$gene_dir': $!.\n" if (!-e $gene_dir); + +# Check if completed genes from a previous run exist +my %complete_genes; +if (-e $mb_archive) { + print "\nArchive containing completed MrBayes runs found for this dataset found in '$mb_archive'.\n"; + print "Completed runs contained in this archive will be removed from the job queue.\n"; + + # Add gene names in tarball to list of completed genes + chomp(my @complete_genes = `tar tf '$mb_archive'`); + foreach my $gene (@complete_genes) { + $gene =~ s/\.tar\.gz//; + $complete_genes{$gene}++; + } +} + +# Unarchive input genes +chomp(my @genes = `tar xvf '$init_dir/$archive' -C $gene_dir 2>&1`); +@genes = map { s/x //; $_ } @genes if ($os_name eq "darwin"); + +chdir($gene_dir); + +# Remove completed genes +if (%complete_genes) { + foreach my $index (reverse(0 .. $#genes)) { + if (exists($complete_genes{$genes[$index]})) { + unlink($genes[$index]); + splice(@genes, $index, 1); + } + } +} + +die "\nAll jobs have already completed.\n\n" if (!@genes); + +# Append given MrBayes block to the end of each gene +print "\nAppending MrBayes block to each gene... "; +foreach my $gene (@genes) { + open(my $gene_file, ">>", $gene) or die "Could not open '$gene': $!.\n"; + print {$gene_file} "\n", @mb_block; + close($gene_file); +} +print "done.\n\n"; + +# Returns the external IP address of this computer +chomp(my $server_ip = `dig +short myip.opendns.com \@resolver1.opendns.com 2>&1`); +if ($server_ip !~ /(?:[0-9]{1,3}\.){3}[0-9]{1,3}/) { + print "Could not determine external IP address, only local clients will be created.\n"; + $server_ip = "127.0.0.1"; +} + +# Initialize a server +my $sock = IO::Socket::INET->new( + LocalPort => $port, + Blocking => 0, + Reuse => 1, + Listen => SOMAXCONN, + Proto => 'tcp') +or die "Could not create server socket: $!.\n"; +$sock->autoflush(1); + +print "Job server successfully created.\n"; + +# Should probably do this earlier +# Determine server hostname and add to machines if none were specified by the user +chomp(my $server_hostname = `hostname`); +if (scalar(@machines) == 0) { + push(@machines, $server_hostname); + $machines{$server_hostname} = "127.0.0.1"; +} +elsif (scalar(@machines) == 1) { + # Check if the user input only the local machine in the config + if ($machines{$machines[0]} eq $server_ip) { + $machines{$machines[0]} = "127.0.0.1"; + } +} + +my @pids; +foreach my $machine (@machines) { + + # Fork and create a client on the given machine + my $pid = fork(); + if ($pid == 0) { + close(STDIN); + close(STDOUT); + close(STDERR); + + (my $script_name = $script_path) =~ s/.*\///; + + # Move required datafiles to machines, initialize clients + if ($machines{$machine} ne "127.0.0.1" && $machines{$machine} ne $server_ip) { + # Send this script to the machine + system("scp", "-q", $script_path, $machine.":/tmp"); + + # Send MrBayes executable to the machine + system("scp", "-q", $mb, $machine.":/tmp"); + + # Execute this perl script on the given machine + # -tt forces pseudo-terminal allocation and lets us stop remote processes + exec("ssh", "-tt", "$machine", "perl", "/tmp/$script_name", "--server-ip=$server_ip:$port"); + } + else { + # Send this script to the machine + system("cp", $script_path, "/tmp"); + + # Send MrBayes executable to the machine + system("cp", $mb, "/tmp"); + + # Execute this perl script on the given machine + exec("perl", "/tmp/$script_name", "--server-ip=127.0.0.1:$port"); + } + + exit(0); + } + else { + push(@pids, $pid); + } +} + +#chdir($gene_dir); + +my $select = IO::Select->new($sock); + +# Don't create zombies +$SIG{CHLD} = 'IGNORE'; + +# Stores which job is next in queue +my $job_number = 0; + +# Number of open connections to a client +my $total_connections; + +# Number of complete jobs (necessary?) +my $complete_count = 0; + +# Number of connections server has closed +my $closed_connections = 0; + +# Minimum number of connections server should expect +my $starting_connections = scalar(@machines); + +my $time = time(); +my $num_digits = get_num_digits({'NUMBER' => scalar(@genes)}); + +# Begin the server's job distribution +while ((!defined($total_connections) || $closed_connections != $total_connections) || $total_connections < $starting_connections) { + # Contains handles to clients which have sent information to the server + my @clients = $select->can_read(0); + + # Free up CPU by sleeping for 10 ms + usleep(10000); + + # Handle each ready client individually + CLIENT: foreach my $client (@clients) { + + # Client requesting new connection + if ($client == $sock) { + $total_connections++; + $select->add($sock->accept()); + } + else { + + # Get client's message + my $response = <$client>; + next if (not defined($response)); # a response should never actually be undefined + + # Client wants to send us a file + if ($response =~ /SEND_FILE: (.*)/) { + my $file_name = $1; + receive_file({'FILE_PATH' => $file_name, 'FILE_HANDLE' => $client}); + } + + # Client has finished a job + if ($response =~ /DONE (.*) \|\|/) { + $complete_count++; + printf(" Analyses complete: %".$num_digits."d/%d.\r", $complete_count, scalar(@genes)); + + # Perform appending of new gene to tarball in a fork as this can take some time + my $pid; + until (defined($pid)) { $pid = fork(); usleep(30000); } + + if ($pid == 0) { + + # Check if this is the first to complete, if so we must create the directory + my $completed_gene = $1; + if (!-e "../$mb_archive") { + system("touch", "$mb_archive"); + system("tar", "cf", "../$mb_archive", $completed_gene); + unlink($completed_gene); + } + else { + + # Obtain a file lock on archive so another process doesn't simultaneously try to add to it + open(my $mb_archive_file, "<", "../$mb_archive"); + flock($mb_archive_file, LOCK_EX) || die "Could not lock '$mb_archive_file': $!.\n"; + + # Add completed gene + system("tar", "rf", "../$mb_archive", $completed_gene); + unlink($completed_gene); + + # Release lock + flock($mb_archive_file, LOCK_UN) || die "Could not unlock '$mb_archive_file': $!.\n"; + close($mb_archive_file); + + } + exit(0); + } + else { + push(@pids, $pid); + } + } + + # Client wants a new job + if ($response =~ /NEW: (.*)/) { + my $client_ip = $1; + + # Check if jobs remain in the queue + if ($job_number < scalar(@genes)) { + printf("\n Analyses complete: %".$num_digits."d/%d.\r", 0, scalar(@genes)) if ($job_number == 0); + + my $gene = $genes[$job_number]; + + # Check whether the client is remote or local, send it needed files if remote + if ($client_ip ne $server_ip) { + + # Fork to perform the file transfer and prevent stalling the server + my $pid; + until (defined($pid)) { $pid = fork(); usleep(30000); } + + #my $pid = fork(); + if ($pid == 0) { + send_file({'FILE_PATH' => $gene, 'FILE_HANDLE' => $client}); + unlink($gene); + + print {$client} "NEW: $gene\n"; + exit(0); + } + else { + push(@pids, $pid); + } + } + else { + print {$client} "CHDIR: ".abs_path("./")."\n"; + print {$client} "NEW: $gene\n"; + } + $job_number++; + } + else { + # Client has asked for a job, but there are none remaining + print {$client} "HANGUP\n"; + $select->remove($client); + $client->close(); + $closed_connections++; + next CLIENT; + } + } + } + } +} + +# Don't think this is needed +foreach my $pid (@pids) { + waitpid($pid, 0); +} + +print "\n All connections closed.\n"; +print "Total execution time: ", sec2human(time() - $time), ".\n\n"; + +# Go back to project directory, delete empty gene dir +chdir(".."); +&INT_handler; + +sub client { + my ($opt_name, $address) = @_; + + my ($server_ip, $port) = split(":", $address); + + chdir("/tmp"); + my $mb = "/tmp/mb"; + + #my $pgrp = getpgrp(); + my $pgrp = $$; + setpgrp(); + + # Determine this host's IP + chomp(my $ip = `dig +short myip.opendns.com \@resolver1.opendns.com`); + + # Set IP to localhost if we don't have internet + if ($ip !~ /(?:[0-9]{1,3}\.){3}[0-9]{1,3}/) { + $ip = "127.0.0.1"; + } + + # Spawn more clients + my @pids; + # my $total_forks = get_free_cpus(); + # A slightly modification in order to limit forks to 10 + my $total_forks = 12; + if ($total_forks > 1) { + foreach my $fork (1 .. $total_forks - 1) { + + my $pid = fork(); + if ($pid == 0) { + last; + } + else { + push(@pids, $pid); + } + } + } + + # The name of the gene we are working on + my $gene; + + # Stores filenames of unneeded files + my @unlink; + + # Change signal handling so killing the server kills these processes and cleans up + $SIG{CHLD} = 'IGNORE'; + $SIG{HUP} = sub { unlink($0, $mb); kill -15, $$; }; + $SIG{TERM} = sub { unlink(glob($gene."*")) if defined($gene); exit(0)}; + + # Connect to the server + my $sock = new IO::Socket::INET( + PeerAddr => $server_ip.":".$port, + Proto => 'tcp') + or exit(0); + $sock->autoflush(1); + + print {$sock} "NEW: $ip\n"; + while (chomp(my $response = <$sock>)) { + + if ($response =~ /SEND_FILE: (.*)/) { + my $file_name = $1; + receive_file({'FILE_PATH' => $file_name, 'FILE_HANDLE' => $sock}); + } + elsif ($response =~ /CHDIR: (.*)/) { + chdir($1); + } + elsif ($response =~ /NEW: (.*)/) { + $gene = $1; + + # Redirect STDOUT to a log file + open(my $std_out, ">&", *STDOUT); + open(STDOUT, ">", $gene.".log"); + + system($mb, $gene); + + # Put STDOUT back to normal + open(STDOUT, ">&", $std_out); + close($std_out); + + unlink($gene); + + # Zip and tarball the results + my @results = glob($gene."*"); + my $gene_archive_name = "$gene.tar.gz"; + @results = grep {!/\Q$gene_archive_name\E/} @results; + system("tar", "czf", $gene_archive_name, @results); + unlink(@results); + + # Send the results back to the server if this is a remote client + if ($server_ip ne "127.0.0.1" && $server_ip ne $ip) { + send_file({'FILE_PATH' => $gene_archive_name, 'FILE_HANDLE' => $sock}); + unlink($gene_archive_name); + } + + # Request a new job + print {$sock} "DONE $gene_archive_name || NEW: $ip\n"; + } + elsif ($response eq "HANGUP") { + last; + } + } + + # Have initial client wait for all others to finish and clean up + if ($$ == $pgrp) { + foreach my $pid (@pids) { + waitpid($pid, 0); + } + unlink($0, $mb); + } + + exit(0); +} + +sub check_nonconvergent { + my ($opt_name, $threshold) = @_; + + # We have to do weird things here to get the input name + + my @ARGV = split(/\s+/, $invocation); + shift(@ARGV); shift(@ARGV); + + # Look for a directory in arguments provided + my $archive; + foreach my $arg (@ARGV) { + if (-d $arg) { + $archive = $arg; + } + } + + # Die if user didn't give us a directory + if (!defined($archive)) { + print "You must specify a directory previously generated by this script to check for nonconvergent genes.\n"; + exit(0); + } + + # Prior output available, set relevant variables + + $project_name = $archive; + my @contents = glob("$project_name/*"); + + # Determine the archive name by looking for a symlink + my $found_name = 0; + foreach my $file (@contents) { + if (-l $file) { + $file =~ s/\Q$project_name\E\///; + $archive = $file; + $found_name = 1; + } + } + die "Could not locate archive in '$project_name'.\n" if (!$found_name); + + chdir($project_name); + + # Extract name information from input file + (my $archive_root = $archive) =~ s/.*\/(.*)/$1/; + (my $archive_root_no_ext = $archive) =~ s/(.*\/)?(.*)(\.tar\.gz)|(\.tgz)/$2/; + + # Should have some completed genes in it + my $incomplete_archive = $archive_root_no_ext.".mb.tar"; + + # Check that the incomplete archive exists + if (!-e $incomplete_archive) { + print "Could not locate an archive containing completed MrBayes runs.\n"; + exit(0); + } + + print "\nScript was called as follows:\n$invocation\n\n"; + + # Create a temporary directory for our operations + my $check_dir = "tmp/"; + mkdir($check_dir) if (!-e $check_dir); + + $SIG{INT} = sub { remove_tree($check_dir); exit(0) }; + + # Open tarball in genes directory + chomp(my @genes = `tar xvf '$incomplete_archive' -C $check_dir 2>&1`); + @genes = map { s/x //; $_ } @genes if ($os_name eq "darwin"); + @genes = sort { (local $a = $a) =~ s/.*-(\d+)-\d+\..*/$1/; + (local $b = $b) =~ s/.*-(\d+)-\d+\..*/$1/; + $a <=> $b } @genes; + my $longest_name_length = length($genes[$#genes]); + + print "MrBayes results available for ", scalar(@genes), " total genes:\n"; + + chdir($check_dir); + + $SIG{INT} = sub { chdir(".."); remove_tree($check_dir); exit(0) }; + + # Parse log of each gene to determine final standard deviation of split frequencies + + my $count = 0; + foreach my $gene (@genes) { + + chomp(my @contents = `tar xvf '$gene' 2>&1`); + @contents = map { s/x //; $_ } @contents if ($os_name eq "darwin"); + + (my $log_file_path = $gene) =~ s/\.tar\.gz$/.log/; + + # Check log file exists + if (!-e $log_file_path) { + print "Could not locate log file for '$gene'.\n"; + exit(0); + } + + open(my $log_file, "<", $log_file_path); + chomp(my @data = <$log_file>); + close($log_file); + + my @splits = grep { /Average standard deviation of split frequencies:/ } @data; + my $final_split = pop(@splits); + + $final_split =~ s/.*frequencies: (.*)/$1/; + + #print " $gene: $final_split\n"; + printf(" %-${longest_name_length}s: %s\n", $gene, $final_split); + + if (!defined($final_split) || $final_split > $threshold) { + $count++; + } + unlink(@contents); + } + printf("%d gene(s) failed to meet the threshold of %s (%.2f%%).\n", $count, $threshold, ($count / scalar(@genes) * 100)); + + # Clean up and exit + kill(2, $$); +} + +sub remove_nonconvergent { + my ($opt_name, $threshold) = @_; + + # We have to do weird things here to get the input name + + my @ARGV = split(/\s+/, $invocation); + shift(@ARGV); shift(@ARGV); + + # Look for a directory in arguments provided + my $archive; + foreach my $arg (@ARGV) { + if (-d $arg) { + $archive = $arg; + } + } + + # Die if user didn't give us a directory + if (!defined($archive)) { + print "You must specify a directory previously generated by this script to check for nonconvergent genes.\n"; + exit(0); + } + + my $initial_archive = $archive; + + # Prior output available, set relevant variables + + $project_name = $archive; + my @contents = glob("$project_name/*"); + + # Determine the archive name by looking for a symlink + my $found_name = 0; + foreach my $file (@contents) { + if (-l $file) { + $file =~ s/\Q$project_name\E\///; + $archive = $file; + $found_name = 1; + } + } + die "Could not locate archive in '$project_name'.\n" if (!$found_name); + + chdir($project_name); + + # Extract name information from input file + (my $archive_root = $archive) =~ s/.*\/(.*)/$1/; + (my $archive_root_no_ext = $archive) =~ s/(.*\/)?(.*)(\.tar\.gz)|(\.tgz)/$2/; + + # Should have some completed genes in it + my $incomplete_archive = $archive_root_no_ext.".mb.tar"; + + # Check that the incomplete archive exists + if (!-e $incomplete_archive) { + print "Could not locate an archive containing completed MrBayes runs.\n"; + exit(0); + } + + print "\nScript was called as follows:\n$invocation\n\n"; + + # Create a temporary directory for our operations + my $check_dir = "tmp/"; + mkdir($check_dir) if (!-e $check_dir); + + $SIG{INT} = sub { remove_tree($check_dir); exit(0) }; + + # Open tarball in genes directory + chomp(my @genes = `tar xvf '$incomplete_archive' -C $check_dir 2>&1`); + @genes = map { s/x //; $_ } @genes if ($os_name eq "darwin"); + @genes = sort { (local $a = $a) =~ s/.*-(\d+)-\d+\..*/$1/; + (local $b = $b) =~ s/.*-(\d+)-\d+\..*/$1/; + $a <=> $b } @genes; + my $longest_name_length = length($genes[$#genes]); + + print "MrBayes results available for ", scalar(@genes), " total genes:\n"; + + chdir($check_dir); + + $SIG{INT} = sub { chdir(".."); remove_tree($check_dir); exit(0) }; + + # Parse log of each gene to determine final standard deviation of split frequencies + + my $count = 0; + foreach my $gene (@genes) { + + chomp(my @contents = `tar xvf '$gene' 2>&1`); + @contents = map { s/x //; $_ } @contents if ($os_name eq "darwin"); + + (my $log_file_path = $gene) =~ s/\.tar\.gz$/.log/; + + # Check log file exists + if (!-e $log_file_path) { + print "Could not locate log file for '$gene'.\n"; + exit(0); + } + + open(my $log_file, "<", $log_file_path); + chomp(my @data = <$log_file>); + close($log_file); + + my @splits = grep { /Average standard deviation of split frequencies:/ } @data; + my $final_split = pop(@splits); + + $final_split =~ s/.*frequencies: (.*)/$1/; + + #print " $gene: $final_split"; + if (!defined($final_split) || $final_split > $threshold) { + unlink($gene); + #print " -- REMOVED\n"; + printf(" %-${longest_name_length}s: %s -- REMOVED\n", $gene, $final_split); + $count++; + } + else { + printf(" %-${longest_name_length}s: %s\n", $gene, $final_split); + #print "\n"; + } + unlink(@contents); + } + printf("%d gene(s) failed to meet the threshold of %s (%.2f%%) and have been removed.\n", $count, $threshold, ($count / scalar(@genes) * 100)); + + # Determine which genes met threshold and still remain + @genes = glob($archive_root_no_ext."*.nex.tar.gz"); + @genes = sort { (local $a = $a) =~ s/.*-(\d+)-\d+\..*/$1/; + (local $b = $b) =~ s/.*-(\d+)-\d+\..*/$1/; + $a <=> $b } @genes; + + # Recreate archive with remaining genes + if (@genes) { + system("tar", "cf", $incomplete_archive, @genes); + unlink(@genes); + system("mv", $incomplete_archive, ".."); + } + else { + # Delete the working directory if no genes meet the threshold + print "No genes met the threshold, removing specified directory.\n"; + + chdir($initial_directory); + $SIG{INT} = sub { remove_tree($initial_archive); exit(0) }; + } + + # Clean up and exit + kill(2, $$); +} + +sub hashsum { + my $settings = shift; + + my $file_path = $settings->{'FILE_PATH'}; + + open(my $file, "<", $file_path) or die "Couldn't open file '$file_path': $!.\n"; + my $md5 = Digest::MD5->new; + my $md5sum = $md5->addfile(*$file)->hexdigest; + close($file); + + return $md5sum; +} + +sub send_file { + my $settings = shift; + + my $file_path = $settings->{'FILE_PATH'}; + my $file_handle = $settings->{'FILE_HANDLE'}; + + my $hash = hashsum({'FILE_PATH' => $file_path}); + print {$file_handle} "SEND_FILE: $file_path\n"; + + open(my $file, "<", $file_path) or die "Couldn't open file '$file_path': $!.\n"; + while (<$file>) { + print {$file_handle} $_; + } + close($file); + + print {$file_handle} " END_FILE: $hash\n"; + + # Stall until we know status of file transfer + while (defined(my $response = <$file_handle>)) { + chomp($response); + + last if ($response eq "TRANSFER_SUCCESS"); + die "Unsuccessful file transfer, checksums did not match.\n" if ($response eq "TRANSFER_FAILURE"); + } +} + +sub receive_file { + my $settings = shift; + + my $file_path = $settings->{'FILE_PATH'}; + my $file_handle = $settings->{'FILE_HANDLE'}; + + my $check_hash; + open(my $file, ">", $file_path); + while (<$file_handle>) { + if ($_ =~ /(.*) END_FILE: (\S+)/) { + print {$file} $1; + $check_hash = $2; + last; + } + else { + print {$file} $_; + } + } + close($file); + + # Use md5 hashsum to make sure transfer worked + my $hash = hashsum({'FILE_PATH' => $file_path}); + if ($hash ne $check_hash) { + die "Unsuccessful file transfer, checksums do not match.\n'$hash' - '$check_hash'\n"; # hopefully this never pops up + print {$file_handle} "TRANSFER_FAILURE\n" + } + + else { + print {$file_handle} "TRANSFER_SUCCESS\n"; + } +} + +sub INT_handler { + + # Kill ssh process(es) spawn by this script + foreach my $pid (@pids) { + #kill(-1, $pid); + kill(15, $pid); + } + + # Move into gene directory + #chdir("$initial_directory"); + chdir("$initial_directory/$project_name"); + + # Try to delete directory five times, if it can't be deleted print an error message + # I've found this method is necessary for analyses performed on AFS drives + my $count = 0; + until (!-e $gene_dir || $count == 5) { + $count++; + + remove_tree($gene_dir, {error => \my $err}); + sleep(1); + } + #logger("Could not clean all files in './$gene_dir/'.") if ($count == 5); + print "Could not clean all files in './$gene_dir/'.\n" if ($count == 5); + + exit(0); +} + +sub clean_up { + my $settings = shift; + + my $remove_dirs = $settings->{'DIRS'}; + my $current_dir = getcwd(); + +# chdir($alignment_root); +# unlink(glob($gene_dir."$alignment_name*")); +# #unlink($server_check_file) if (defined($server_check_file)); +# +# if ($remove_dirs) { +# rmdir($gene_dir); +# } + chdir($current_dir); +} + +sub get_num_digits { + my $settings = shift; + + my $number = $settings->{'NUMBER'}; + + my $digits = 1; + while (floor($number / 10) != 0) { + $number = floor($number / 10); + $digits++; + } + + return $digits; +} + +sub sec2human { + my $secs = shift; + + # Constants + my $secs_in_min = 60; + my $secs_in_hour = 60 * 60; + my $secs_in_day = 24 * 60 * 60; + + $secs = int($secs); + + return "0 seconds" if (!$secs); + + # Calculate units of time + my $days = int($secs / $secs_in_day); + my $hours = ($secs / $secs_in_hour) % 24; + my $mins = ($secs / $secs_in_min) % 60; + $secs = $secs % 60; + + # Format return nicely + my $time; + if ($days) { + $time .= ($days != 1) ? "$days days, " : "$days day, "; + } + if ($hours) { + $time .= ($hours != 1) ? "$hours hours, " : "$hours hour, "; + } + if ($mins) { + $time .= ($mins != 1) ? "$mins minutes, " : "$mins minute, "; + } + if ($secs) { + $time .= ($secs != 1) ? "$secs seconds " : "$secs second "; + } + else { + # Remove comma + chop($time); + } + chop($time); + + return $time; +} + +sub get_free_cpus { + + return $max_forks if (defined($max_forks)); + + my $os_name = $^O; + + # Returns a two-member array containing CPU usage observed by top, + # top is run twice as its first output is usually inaccurate + my @percent_free_cpu; + if ($os_name eq "darwin") { + # Mac OS + chomp(@percent_free_cpu = `top -i 1 -l 2 | grep "CPU usage"`); + } + else { + # Linux + chomp(@percent_free_cpu = `top -b -n2 -d0.05 | grep "Cpu(s)"`); + } + + my $percent_free_cpu = pop(@percent_free_cpu); + + if ($os_name eq "darwin") { + # Mac OS + $percent_free_cpu =~ s/.*?(\d+\.\d+)%\s+id.*/$1/; + } + else { + # linux + $percent_free_cpu =~ s/.*?(\d+\.\d)\s*%?ni,\s*(\d+\.\d)\s*%?id.*/$1 + $2/; # also includes %nice as free + $percent_free_cpu = eval($percent_free_cpu); + } + + my $total_cpus; + if ($os_name eq "darwin") { + # Mac OS + $total_cpus = `sysctl -n hw.ncpu`; + } + else { + # linux + $total_cpus = `grep --count 'cpu' /proc/stat` - 1; + } + + my $free_cpus = ceil($total_cpus * $percent_free_cpu / 100); + + if ($free_cpus == 0 || $free_cpus !~ /^\d+$/) { + $free_cpus = 1; # assume that at least one cpu can be used + } + + return $free_cpus; +} + +sub run_cmd { + my $command = shift; + + my $return = system($command); + + if ($return) { + logger("'$command' died with error: '$return'.\n"); + #kill(2, $parent_pid); + exit(0); + } +} + +sub check_path_for_exec { + my $exec = shift; + + my $path = $ENV{PATH}.":."; # include current directory as well + my @path_dirs = split(":", $path); + + my $exec_path; + foreach my $dir (@path_dirs) { + $dir .= "/" if ($dir !~ /\/$/); + $exec_path = abs_path($dir.$exec) if (-e $dir.$exec && -x $dir.$exec && !-d $dir.$exec); + } + + die "Could not find the following executable: '$exec'. This script requires this program in your path.\n" if (!defined($exec_path)); + return $exec_path; +} + +sub usage { + return "Usage: mb.pl ([PARTITION TARBALL] [-m MRBAYES BLOCK]) || ([MRBAYES TARBALL] [-c THRESHOLD] || [-r THRESHOLD])\n"; +} + +sub help { +print < +EOF +exit(0); +} diff --git a/99.scripts/trinity_utils/PerlLib/Ascii_genome_illustrator.pm b/99.scripts/trinity_utils/PerlLib/Ascii_genome_illustrator.pm new file mode 100644 index 0000000..efa2cf9 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Ascii_genome_illustrator.pm @@ -0,0 +1,157 @@ +package Ascii_genome_illustrator; + +use strict; +use warnings; +use Carp; + +sub new { + my $packagename = shift; + + my ($molecule_name, $illustration_length) = @_; + + my $self = { name => $molecule_name, + illustration_length => $illustration_length, + features => [], + }; + + bless ($self, $packagename); + + return ($self); +} + + +#### +sub add_features { ## accepts list of features + my $self = shift; + my @features = @_; + + foreach my $feature (@features) { + unless (ref $feature eq 'ARRAY') { confess "Error, improper params; should be a list of feature array refs"; } + $self->add_feature(@$feature); + } + + + return; +} + +#### +sub add_feature { + my $self = shift; + my ($feature_name, $feature_end5, $feature_end3, $glyph) = @_; + + unless ($feature_name && $feature_end5 =~ /^\d+$/ && $feature_end3 =~ /^\d+$/ && defined($glyph) ) { + confess "Error, improper params"; + } + + unless (length($glyph) == 1 && $glyph !~ /\s/) { confess "Error, glyph must be a single non ws character.";} + + my $orient = ($feature_end5 < $feature_end3) ? '+' : '-'; + + my ($lend, $rend) = sort {$a<=>$b} ($feature_end5, $feature_end3); + + #print "Feature: $feature_name, $lend => $rend ($orient)\n"; + + my $feature_struct = { name => $feature_name, + lend => $lend, + rend => $rend, + orient => $orient, + glyph => $glyph, + }; + + push (@{$self->{features}}, $feature_struct); + + return; +} + + +#### +sub get_features { + my $self = shift; + return (@{$self->{features}}); +} + + + +#### +sub illustrate { + my $self = shift; + + my ($mol_lend, $mol_rend) = @_; + + + unless ($mol_lend && $mol_rend) { confess "invalid params"; } + + my $illustration_length = $self->{illustration_length}; + + my $text = sprintf("%+20s ", "$mol_lend-$mol_rend") . "[" . ("=" x ($illustration_length-2)) . "]\t" . $self->{name} . "\n"; + + foreach my $feature ($self->get_features()) { + + my ($name, $lend, $rend, $orient, $glyph) = ($feature->{name}, $feature->{lend}, $feature->{rend}, $feature->{orient}, $feature->{glyph}); + + my @illustration_array; + ## init to clean palette + for (my $i = 0; $i < $illustration_length; $i++) { $illustration_array[$i] = " "; } + + my ($pos_lend, $pos_rend) = $self->_compute_palette_position([$lend, $rend], [$mol_lend, $mol_rend]); + + # draw + for (my $i = $pos_lend; $i <= $pos_rend; $i++) { $illustration_array[$i] = $glyph; } + + if ($orient eq '+') { + $illustration_array[$pos_rend] = '>'; + } + elsif ($orient eq '-') { + $illustration_array[$pos_lend] = '<'; + } + else { + confess "Don't recognize orient:$orient"; + } + + $text .= sprintf ("%+20s$orient ", "$lend-$rend") . join ("", @illustration_array) . "\t$name\n"; + } + + return ($text); +} + + +#### +sub _compute_palette_position { + my $self = shift; + + my ($feature_coords_aref, $mol_coords_aref) = @_; + + my $illustration_length = $self->{illustration_length}; + + my ($feature_lend, $feature_rend) = @$feature_coords_aref; + my ($mol_lend, $mol_rend) = @$mol_coords_aref; + + unless ($feature_lend <= $mol_rend && $feature_rend >= $mol_lend) { + ## no overlap + return (-1, -1); + } + + if ($feature_lend < $mol_lend) { + $feature_lend = $mol_lend; + } + + if ($feature_rend > $mol_rend) { + $feature_rend = $mol_rend; + } + + my $mol_region_length = $mol_rend - $mol_lend + 1; + + my $delta_left = $feature_lend - $mol_lend + 1; + my $delta_right = $feature_rend - $mol_lend + 1; + + my $pos_left = int($delta_left / $mol_region_length * $illustration_length + 0.5) - 1; + $pos_left = 0 if $pos_left < 0; + my $pos_right = int($delta_right / $mol_region_length * $illustration_length + 0.5) - 1; + $pos_right = 0 if $pos_right < 0; + + return ($pos_left, $pos_right); +} + + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/BED_utils.pm b/99.scripts/trinity_utils/PerlLib/BED_utils.pm new file mode 100644 index 0000000..a9c143f --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/BED_utils.pm @@ -0,0 +1,54 @@ +package BED_utils; + +use strict; +use warnings; +use Carp; +use Gene_obj; + +sub index_BED_as_gene_objs { + my ($gff_filename, $gene_id_to_gene_obj_href) = @_; + + my %contig_to_gene_list; + + open (my $fh, $gff_filename) or die "Error, cannot open file $gff_filename"; + while (<$fh>) { + if (/^\#/) { next; } + chomp; + unless (/\w/) { next; } + + my $bed_line = $_; + + my $gene_obj; + + eval { + $gene_obj = &Gene_obj::BED_line_to_gene_obj($bed_line); + my @introns = $gene_obj->get_intron_coordinates(); # this method breaks if all exons are single bases. Ignore these weird things. + }; + + if ($@) { + print STDERR "ERROR, cannot create gene for bed line:\n$bed_line\n$@\n"; + next; + } + + + my $gene_id = $gene_obj->{TU_feat_name}; + + my $indexed_gene_obj = $gene_id_to_gene_obj_href->{$gene_id}; + if ($indexed_gene_obj) { + $indexed_gene_obj->add_isoform($gene_obj); + } + else { + $gene_id_to_gene_obj_href->{$gene_id} = $gene_obj; + my $contig = $gene_obj->{asmbl_id}; + push (@{$contig_to_gene_list{$contig}}, $gene_id); + } + } + close $fh; + + return(\%contig_to_gene_list); +} + + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/BHStats.pm b/99.scripts/trinity_utils/PerlLib/BHStats.pm new file mode 100644 index 0000000..b619bd3 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/BHStats.pm @@ -0,0 +1,317 @@ +package BHStats; + +require Exporter; +our @ISA = qw (Exporter); +our @EXPORT = qw (binomial_probability_sum_k_to_n + binomial_probability_sum_k_to_0 + binomial_probability + binomial_coefficient + factorial + stDev + stdErr + median + avg + CorrelationCoeff + geometric_mean + min + max + sum + tukey_biweight + ); + + +use strict; + +sub binomial_probability_sum_k_to_n { + my ($n,$k,$p) = @_; + my $sum = 0; + for (my $i = $k; $i <= $n; $i++) { + my $binProb = binomial_probability($n,$i,$p); + $sum += $binProb; + } + return ($sum); +} + +sub binomial_probability_sum_k_to_0 { + my ($n,$k,$p) = @_; + my $sum = 0; + for (my $i = $k; $i >= 0; $i--) { + my $binProb = binomial_probability($n,$i,$p); + $sum += $binProb; + } + return ($sum); +} + + + + +sub binomial_probability { + my ($n_observations, $k_successes, $p_probability) = @_; + + my ($n, $k, $p) = ($n_observations, $k_successes, $p_probability); + + ### Given B(n,p), find P(X=k) + + my $binomial_prob = binomial_coefficient($n,$k) * ($p**$k) * (1-$p)**($n-$k); + + return ($binomial_prob); + +} + + +sub binomial_coefficient { + my ($n_things, $k_at_a_time) = @_; + + my $number_of_k_arrangements = (factorial($n_things)) / ( factorial($k_at_a_time) * factorial($n_things-$k_at_a_time) ); + + return ($number_of_k_arrangements); +} + + +sub factorial { + my $x = shift; + $x = int($x); + my $factorial = 1; + while ($x > 1) { + $factorial *= $x; + $x--; + } + return ($factorial); +} + + +sub stDev { + # standard deviation calculation + my @nums = @_; + @nums = sort {$a<=>$b} @nums; + + + my $avg = avg(@nums); + my $count_eles = scalar(@nums); + + ## sum up the sqr of diff from avg + my $sum_avg_diffs_sqr = 0; + foreach my $num (@nums) { + my $diff = $num - $avg; + my $sqr = $diff**2; + $sum_avg_diffs_sqr += $sqr; + } + my $stdev = sqrt ($sum_avg_diffs_sqr/($count_eles-1)); + return ($stdev); +} + +#### +sub stdErr { + my @vals = @_; + + my $stdev = &stDev(@vals); + + my $num_vals = scalar(@vals); + + my $stdErr = $stdev / sqrt($num_vals); + + return($stdErr); +} + + +sub median { + my @nums = @_; + + @nums = sort {$a<=>$b} @nums; + + my $count = scalar (@nums); + if ($count %2 == 0) { + ## even number: + my $half = $count / 2; + return (avg ($nums[$half-1], $nums[$half])); + } + else { + ## odd number. Return middle value + my $middle_index = int($count/2); + return ($nums[$middle_index]); + } +} + +sub avg { + my @nums = @_; + my $total = $#nums + 1; + my $sum = 0; + foreach my $num (@nums) { + $sum += $num; + } + my $avg = $sum/$total; + return ($avg); +} + + +sub CorrelationCoeff { + my ($x_aref, $y_aref) = @_; + my @x = @$x_aref; + my @y = @$y_aref; + + my $total = $#x + 1; + my $avg_x = avg(@x); + my $avg_y = avg(@y); + + my $stdev_x = stDev(@x); + my $stdev_y = stDev(@y); + + # sum part of equation + my $summation = 0; + for (my $i = 0; $i < $total; $i++) { + my $x_val = $x[$i]; + my $y_val = $y[$i]; + + my $x_part = ($x_val - $avg_x)/$stdev_x; + my $y_part = ($y_val - $avg_y)/$stdev_y; + + $summation += ($x_part * $y_part); + } + + my $cor = (1/($total-1)) * $summation; + + return ($cor); +} + + +#### +sub geometric_mean { + my @entries = @_; + + my $num_entries = scalar (@entries); + unless ($num_entries) { + return (undef); + } + + ## All entries must be > 0 + my $logsum = 0; + foreach my $entry (@entries) { + unless ($entry > 0) { + return (undef); + } + $logsum += log ($entry); + } + + my $geo_mean = exp ( (1/$num_entries) * $logsum); + + return ($geo_mean); +} + + +#### +sub min { + my @vals = @_; + + @vals = sort {$a<=>$b} @vals; + + my $min_val = shift @vals; + + return ($min_val); +} + +#### +sub max { + my @vals = @_; + + @vals = sort {$a<=>$b} @vals; + + my $max_val = pop @vals; + + return ($max_val); +} + + +#### +sub sum { + my @vals = @_; + + my $x = 0; + foreach my $val (@vals) { + $x += $val; + } + + return ($x); +} + + +=Rcode + +> tukey.biweight +function (x, c = 5, epsilon = 1e-04) +{ + m <- median(x) + s <- median(abs(x - m)) + u <- (x - m)/(c * s + epsilon) + w <- rep(0, length(x)) + i <- abs(u) <= 1 + w[i] <- ((1 - u^2)^2)[i] + t.bi <- sum(w * x)/sum(w) + return(t.bi) +} +=cut + +#### +sub tukey_biweight { + my (@x) = @_; + + my $m = median(@x); + + my $s; + { + my @y; + foreach my $val (@x) { + my $t = abs($val - $m); + push (@y, $t); + } + $s = median(@y); + } + + + my @u; + { + my $epsilon = 1e-4; + my $c = 5; + + foreach my $val (@x) { + my $t = $val - $m; + $t /= ($c * $s + $epsilon); + push (@u, $t); + } + } + + my @i; + { + foreach my $val (@u) { + my $t = (abs($val) <= 1) ? 1:0; + push (@i, $t); + } + } + + my @w; + { + foreach my $val (@u) { + my $i = shift @i; + my $t = ( (1 - $val**2) **2) * $i; + push (@w, $t); + } + } + + my $bi; + { + my $sum_w = 0; + for (my $i = 0; $i < @w; $i++) { + my $w = $w[$i]; + my $x = $x[$i]; + $bi += $w * $x; + $sum_w += $w; + } + $bi /= $sum_w; + } + + return($bi); +} + + + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/CDNA/Alignment_segment.pm b/99.scripts/trinity_utils/PerlLib/CDNA/Alignment_segment.pm new file mode 100644 index 0000000..314d69f --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/CDNA/Alignment_segment.pm @@ -0,0 +1,465 @@ + +########################### +### Class Alignment_segment +########################### + +=head1 NAME + +package CDNA::Alignment_segment + +=head1 DESCRIPTION + +Provides an object representation of alignment segments which are built into a single CDNA_alignment object. + +=cut + + + +package CDNA::Alignment_segment; +use strict; +use Data::Dumper; + + +=over 4 + +=item new() + +B Instantiates a new Alignment_segment object. + +B $genomic_end5, $genomic_end3, $cdna_end5, $cdna_end3, $per_id + +B Alignment_segment_obj + +Alignment_segment_obj is an object of type CDNA::Alignment_object + +Use the methods described below. In addition, the following fields are supported: + +B (orientation of the alignment segment) + +B (left end of the alignment segment corresponding to the genomic sequence) + +B (right end of the alignmetn segment corresponding to the genomic sequence) + +note: lend <= rend in all cases; must use the B field to determine cDNA alignment orientation for the segment. + +B (the cDNA coordinate corresponding to the lend alignment coordinate) + +B (the cDNA coordinate corresponding to the rend alignment coordinate) + +=back + +=cut + + +sub new { + my $packagename = shift; + my ($genomic_end5, $genomic_end3, $cdna_end5, $cdna_end3, $per_id) = @_; + my $orientation = '?'; #initialize + my ($lend, $rend, $mlend, $mrend) = ($genomic_end5, $genomic_end3, $cdna_end5, $cdna_end3); + + #reorient coordsets so that cdna coordinates are always in forward orientation. + if ($cdna_end5 > $cdna_end3) { #swap coordsets + ($lend, $rend) = ($rend, $lend); + ($mlend, $mrend) = ($mrend, $mlend); + } + + ## Check orientation and adjust lend, rend accordingly. + if ($lend > $rend) { + $orientation = '-'; + ($lend, $rend) = ($rend, $lend); + ($mlend, $mrend) = ($mrend, $mlend); + } elsif ($lend < $rend) { #keep coords way they are. + $orientation = '+'; + } + + my $self = { + orientation=>$orientation, ## should be [+-] + lend=>$lend, + rend=>$rend, + mlend=>$mlend, ## cDNA coordinate that maps to lend of alignment. + mrend=>$mrend, + per_id => $per_id, + type=>undef(), # [first|last|internal|single] + has_left_splice_junction=>0, #flag indicating whether the consensus is present. + has_right_splice_junction=>0, + left_splice_site_chars=>undef(), #store the two characters at that splice junction. + right_splice_site_chars=>undef() + }; + bless ($self, $packagename); + return ($self); +} + +sub set_coords { + my $self = shift; + my ($c1, $c2) = @_; + ($c1, $c2) = sort {$a<=>$b} ($c1, $c2); + $self->{lend} = $c1; + $self->{rend} = $c2; +} + + +#### +sub get_aligned_orientation { + my $self = shift; + return ($self->{orientation}); +} + + + +=over 4 + +=item get_coords() + +B Retrieves the lend, rend for the alignment segment. + +B none. + +B ($lend, $rend) + +=back + +=cut + + +sub get_coords { + my $self = shift; + return ($self->{lend}, $self->{rend}); +} + +# private. +sub set_mcoords () { + my $self = shift; + my ($mlend, $mrend) = @_; + $self->{mlend} = $mlend; + $self->{mrend} = $mrend; +} + +=over 4 + +=item get_mcoords() + +B Retrieves the mlend, mrend for the cDNA coordinates. + +B none + +B ($mlend, $mrend) + +=back + +=cut + +sub get_mcoords () { + my $self = shift; + return ($self->{mlend}, $self->{mrend}); +} + + +=over 4 + +=item get_per_id() + +B Retrieves the per_id for the alignment segment + +B none + +B $per_id + +=back + +=cut + +sub get_per_id { + my $self = shift; + return ($self->{per_id}); +} + + + +#private +sub set_orientation { + my $self = shift; + my $orientation = shift; + $self->{orientation} = $orientation; +} + +=over 4 + +=item get_orientation() + +B Retrieves the orientation for an alignment segment. + +B none + +B [+|-] + +=back + +=cut + +sub get_orientation { + my $self = shift; + return ($self->{orientation}); +} + +sub set_type { + my $self = shift; + my $type = shift; + unless ($type =~ /first|last|internal|single/) { + die "Incompatible segment type provided: $type\n"; + } + $self->{type} = $type; +} + + +=over 4 + +=item get_type() + +BRetrieves the classification of the alignment segment + +B none + +B [first|last|internal|single] + +=back + +=cut + +sub get_type { + my $self = shift; + return ($self->{type}); +} + +sub is_first { + my $self = shift; + return ($self->{type} eq "first") ; +} + +sub is_internal { + my $self = shift; + return ($self->{type} eq "internal"); +} + +sub is_last { + my $self = shift; + return ($self->{type} eq "last"); +} + +sub is_single_segment { + my $self = shift; + return ($self->{type} eq "single"); +} + +sub set_left_splice_junction { + my $self = shift; + my $value = shift; + $self->{has_left_splice_junction} = $value; +} + +=over 4 + +=item has_left_splice_junction() + +B Provides result of a left splice junction test. + +B none + +B [1|0] + +1=true + +0=false + +=back + +=cut + + +sub has_left_splice_junction { + my $self = shift; + return ($self->{has_left_splice_junction}); +} + +sub set_right_splice_junction { + my $self = shift; + my $value = shift; + $self->{has_right_splice_junction} = $value; +} + +=over 4 + +=item has_right_splice_junction() + +B Provides the result of a right splice junction test. + +B none + +B [1|0] + +=back + +=cut + + +sub has_right_splice_junction { + my $self = shift; + return ($self->{has_right_splice_junction}); +} + +sub set_left_splice_site_chars () { + my $self = shift; + my $chars = shift; + $self->{left_splice_site_chars} = $chars; +} + + +=over 4 + +=item get_left_splice_site_chars() + +B Retrieves the two characters representing the left splice site + +B none. + +B $twochars + +ie. Typically, this will return AG or AC depending on the spliced orientation. + +=back + +=cut + +sub get_left_splice_site_chars () { + my $self = shift; + return ($self->{left_splice_site_chars}); +} + + +sub set_right_splice_site_chars() { + my $self = shift; + my $chars = shift; + $self->{right_splice_site_chars} = $chars; +} + +=over 4 + +=item get_right_splice_chars() + +B Retrieves the two characters representing the right splice site + +B none. + +B $two_chars + +ie. typcially returns GT or CT depending on the spliced orientation. + +=back + +=cut + + +sub get_right_splice_site_chars() { + my $self = shift; + return ($self->{right_splice_site_chars}); +} + +sub toString() { + my $self = shift; + return( "segment\* orient: " . $self->{orientation} . " coords: " . $self->{lend} . "-" . $self->{rend} . " type: " . $self->{type} + . " lsplice: " . $self->has_left_splice_junction() . " rsplice: " . $self->has_right_splice_junction() . "\n"); + +} + +sub toToken() { + my $segment = shift; + my $token = ""; + my $type = $segment->get_type(); + my ($lend, $rend) = $segment->get_coords(); + my ($mlend, $mrend) = $segment->get_mcoords(); + if ($type =~ /internal|last/) { + # check splice site + my $left_splice = $segment->get_left_splice_site_chars(); + if ($segment->has_left_splice_junction()) { + $left_splice = uc $left_splice; + } else { + $left_splice = lc $left_splice; + } + $token .= $left_splice . "<"; + } + $token .= $lend; + if ($mlend) { + $token .= "($mlend)"; + } + $token .= "-$rend"; + if ($rend) { + $token .= "($mrend)"; + } + if ($type =~ /internal|first/) { + # check splice site + my $right_splice = $segment->get_right_splice_site_chars(); + if ($segment->has_right_splice_junction()) { + $right_splice = uc $right_splice; + } else { + $right_splice = lc $right_splice; + } + $token .= ">$right_splice"; + } + return ($token); +} + + + +=over 4 + +=item clone() + +B Clones an Alignment_segment object into a new Alignment_segment object with the same attribute values. + +B none + +B new CDNA::Alignment_segment + +=back + +=cut + + +sub clone { + my $self = shift; + my $packagename = ref $self; + my $clone = {}; + foreach my $key (keys %$self) { + $clone->{$key} = $self->{$key}; + } + bless ($clone, $packagename); + return ($clone); +} + + +=over 4 + +=item get_length() + +B Calculates the length of the alignment segment. + +B none + +B int + +=back + +=cut + + + +sub get_length { + my $self = shift; + my ($lend, $rend) = $self->get_coords(); + my $length = abs ($rend - $lend) + 1; + return ($length); +} + + + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/CDNA/Alternative_splice_comparer.pm b/99.scripts/trinity_utils/PerlLib/CDNA/Alternative_splice_comparer.pm new file mode 100644 index 0000000..cd96b50 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/CDNA/Alternative_splice_comparer.pm @@ -0,0 +1,1000 @@ +#!/usr/local/bin/perl + +package main; +our $SEE; + + +package CDNA::Alternative_splice_comparer; +use Gene_obj; +use strict; +use Data::Dumper; +use Carp; +use CDNA::PASA_alignment_assembler; + +sub new { + my $packagename = shift; + my $self = { + unspliced_introns => 0, + conventional_alt_splice => 0, + start_or_end_within_intron => 0, + exon_skipping => 0, + alternate_exons => 0 + }; + bless ($self, $packagename); + return ($self); +} + + + +#### +sub compare_isoforms_via_alignmentObjs { + my $self = shift; + my ($align1, $align2) = @_; + my $gene_1 = $align1->get_gene_obj_via_alignment(); + my $gene_2 = $align2->get_gene_obj_via_alignment(); + + return ($self->compare_isoforms_via_geneObjs($gene_1, $gene_2)); +} + + +#### +sub compare_isoforms_via_geneObjs { + my $self = shift; + my ($gene1, $gene2) = @_; + ## Looking for: + # -unspliced introns + # -conventional alt-splice isoforms + # -transcriptional start or polyadenylation site within intron + # -exon skipping + # -alternate exons + + ## Look for Unspliced Introns + my $struct = { unspliced_introns => 0, + conventional_alt_splice => 0, + start_or_end_within_intron => 0, + exon_skipping => 0, + alternate_exons=> 0 }; + + + my @unspliced_introns = ($self->find_unspliced_introns($gene1, $gene2), $self->find_unspliced_introns($gene2, $gene1)); + if (@unspliced_introns) { + print "*** Unspliced introns \n"; + $struct->{unspliced_introns} = 1; + $self->{unspliced_introns} = \@unspliced_introns; + } + + + ## Look for the conventional alt splice isoforms (diff donors/acceptors for introns). + my (%alternate_acceptors_n_donors) = $self->find_conventional_alt_splice_isoforms($gene1, $gene2); + if (%alternate_acceptors_n_donors) { + print "*** Conventional Alt splice (donor and/or acceptor)\n"; + #print Dumper (\%alternate_acceptors_n_donors); + + $self->{conventional_alt_splice} = \%alternate_acceptors_n_donors; + if (@{$alternate_acceptors_n_donors{acceptors}}) { + $struct->{conventional_alt_acceptor} = 1; + } + if (@{$alternate_acceptors_n_donors{donors}}) { + $struct->{conventional_alt_donor} = 1; + } + } + + + my %intron_starts_or_ends = ($self->find_starts_and_ends_within_introns ($gene1, $gene2), $self->find_starts_and_ends_within_introns ($gene2, $gene1)); + if (%intron_starts_or_ends) { + print "*** Transcriptional start or polyadenylation site within an intron.\n"; + $struct->{start_or_end_within_intron} = 1; + my @starts_or_ends = values %intron_starts_or_ends; + $self->{start_or_end_within_intron} = \@starts_or_ends; + } + + ## Look for exon skipping events. + my @exon_skips = ($self->find_exon_skipping_events($gene1, $gene2), $self->find_exon_skipping_events($gene2, $gene1)); + if (@exon_skips) { + print "*** Exon skipping event detected.\n"; + $struct->{exon_skipping} = 1; + $self->{exon_skipping} = \@exon_skips; + } + + ## Look or Alternate exons + my @alternate_exons = ($self->find_alternate_exons($gene1, $gene2), $self->find_alternate_exons($gene2, $gene1)); + if (@alternate_exons) { + print "*** Found alternate exons\n"; + $struct->{alternate_exons} = 1; + $self->{alternate_exons} = \@alternate_exons; + } + + return ($struct); + +} + + +# private +sub enumerate_exons_of_gene { + my $gene_obj = shift; + # put everything in forward coordinate axis: + my %exon_coords; + my @exons = $gene_obj->get_exons(); + foreach my $exon (@exons) { + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + $exon_coords{$lend} = $rend; + } + return (%exon_coords); +} + + +#private +#### +sub enumerate_introns_of_gene { + my $gene_obj = shift; + ## Put everything in forward strand coordinate axis. + my %introns; + my @exons = sort {$a->{end5}<=>$b->{end5}} $gene_obj->get_exons(); + for (my $i = 0; $i < $#exons; $i++) { + my ($exon1_lend, $exon1_rend) = sort {$a<=>$b} $exons[$i]->get_coords(); + my ($exon2_lend, $exon2_rend) = sort {$a<=>$b} $exons[$i+1]->get_coords(); + my ($intron_end5, $intron_end3) = ($exon1_rend + 1, $exon2_lend - 1); + $introns{$intron_end5} = $intron_end3; + } + return (%introns); +} + + +=over 4 + +=item find_unspliced_introns() + +B Find unspliced introns in gene_1 when compared to gene_2 + +B $gene1, $gene2 + +B @unspliced_introns + +@unspliced_introns is a list of coordinate pairs representing the unspliced introns found in gene 1 when compared to gene2 + +@unspliced_introns = ([end5,end3], ...) + +=back + +=cut + +#### +sub find_unspliced_introns { + my $self = shift; + my ($gene1, $gene2) = @_; + ## Look for unspliced intron found in gene1 when compared to gene2 + my %gene1_exon_coords = &enumerate_exons_of_gene ($gene1); + my %gene2_intron_coords = &enumerate_introns_of_gene($gene2); + + my @unspliced_introns; + foreach my $intron_lend (keys %gene2_intron_coords) { + my $intron_rend = $gene2_intron_coords{$intron_lend}; + + foreach my $exon_lend (keys %gene1_exon_coords) { + my $exon_rend = $gene1_exon_coords{$exon_lend}; + + if ($intron_lend > $exon_lend && $intron_rend < $exon_rend) { #unspliced intron found + push (@unspliced_introns, [$intron_lend, $intron_rend]); + } + } + } + return (@unspliced_introns); +} + + + +=over 4 + +=item find_conventional_alt_splice_isoforms() + +B Looks for different donor and acceptor sites within overlapping introns of genes + +B $gene1, $gene2 + +B %alt_donors_and_acceptors + +with structure: + +%alt_donors_and_acceptors = ( acceptors => + [ + { gene1 => acceptor_coord, gene2 => acceptor_coord }, ... + + + + ], + + donors => [ + + { gene1 => donor_coord, gene2 => donor_coord }, ... + + + ] + + ); + + Coordinates stored are the actual exon boundary coordinates (first or last bp of each exon) + + + + +=back + +=cut + + +#### +sub find_conventional_alt_splice_isoforms { + my $self = shift; + my ($gene1, $gene2) = @_; + print "## Looking for conventional alt splice isoforms (diff donors, acceptors)\n" if $SEE; + my %exons_1_hash = &enumerate_exons_of_gene ($gene1); + my %exons_2_hash = &enumerate_exons_of_gene ($gene2); + + my $orientation = $gene1->get_orientation(); + if ($orientation ne $gene2->get_orientation()) { + die "Error, inconsistent orientations between genes: " . $gene1->toString() . $gene2->toString(); + } + + # algorithm + # -find one-to-one mappings between exons, and locate differences at acceptors and donor sites. + + my %alternate_acceptors_n_donors = ( acceptors => [], + donors => [] + ); # holds coordinates for all gene1 diff boundaries. + + my $found_diff_flag = 0; + + + ## compare exons of gene1 to gene2 + my @exons_gene_1_list; + my @exons_gene_2_list; + # build data structure: + foreach my $data_pair ( [\%exons_1_hash, \@exons_gene_1_list], + [\%exons_2_hash, \@exons_gene_2_list] ) { + + my ($exons_href, $exons_aref) = @$data_pair; + + foreach my $lend (keys %$exons_href) { + my $rend = $exons_href->{$lend}; + + push (@$exons_aref, { lend => $lend, + rend => $rend, + match_indices => [] } ); + } + } + + @exons_gene_1_list = sort {$a->{lend}<=>$b->{lend}} @exons_gene_1_list; + @exons_gene_2_list = sort {$a->{lend}<=>$b->{lend}} @exons_gene_2_list; + + + # all-vs-all comparison: + for (my $i = 0; $i <= $#exons_gene_1_list; $i++) { + + my $i_ele_ref = $exons_gene_1_list[$i]; + my ($i_lend, $i_rend, $i_match_indices_aref) = ($i_ele_ref->{lend}, + $i_ele_ref->{rend}, + $i_ele_ref->{match_indices} ); + + + for (my $j = 0; $j <= $#exons_gene_2_list; $j++) { + + my $j_ele_ref = $exons_gene_2_list[$j]; + my ($j_lend, $j_rend, $j_match_indices_aref) = ($j_ele_ref->{lend}, + $j_ele_ref->{rend}, + $j_ele_ref->{match_indices}); + + if ($i_lend < $j_rend && $i_rend > $j_lend) { #overlap + push (@$i_match_indices_aref, $j); + push (@$j_match_indices_aref, $i); + } + } + } + + ## find donors and acceptors: + ## check gene_1's exons for 1-1 mappings and end differences at splice junctions + + for (my $i = 0; $i <= $#exons_gene_1_list; $i++) { + + my $i_ele_ref = $exons_gene_1_list[$i]; + my ($i_lend, $i_rend, $i_match_indices_aref) = ($i_ele_ref->{lend}, + $i_ele_ref->{rend}, + $i_ele_ref->{match_indices} ); + + if (scalar (@$i_match_indices_aref) == 1) { + ## found some mapping + my $j_index = $i_match_indices_aref->[0]; + my $j_ele_ref = $exons_gene_2_list[$j_index]; + my ($j_lend, $j_rend, $j_match_indices_aref) = ($j_ele_ref->{lend}, + $j_ele_ref->{rend}, + $j_ele_ref->{match_indices}); + + + if (scalar (@$j_match_indices_aref) != 1) { + next; ## this is a 1-many mapping, want only 1-1 mappings + } + + # make sure j's i is i + if ($j_match_indices_aref->[0] != $i) { + ## bad, this should never happen! + confess "Error, found exon 1-1 mapping of $i to $j_index, but j maps to @$j_match_indices_aref "; + } + + + ## check left boundary: + if ($i_lend != $j_lend ## diff coordinate + && $i != 0 # at a splice junction + && $j_index != 0 # at a splice junction + ) { + + ## found splice difference at left junction: + $found_diff_flag = 1; + + my $splice_ref = ($orientation eq '+') + ? $alternate_acceptors_n_donors{acceptors} + : $alternate_acceptors_n_donors{donors}; + + push (@$splice_ref, { gene1 => $i_lend, + gene2 => $j_lend } ); + + } + + + ## check right boundary: + if ($i_rend != $j_rend ## diff coordinate + && $i != $#exons_gene_1_list # at splice junction + && $j_index != $#exons_gene_2_list # at splice junction + ) { + + $found_diff_flag = 1; + + my $splice_ref = ($orientation eq '+') + ? $alternate_acceptors_n_donors{donors} + : $alternate_acceptors_n_donors{acceptors}; + + push (@$splice_ref, { gene1 => $i_rend, + gene2 => $j_rend } ); + } + } + } + + if ($found_diff_flag) { + return (%alternate_acceptors_n_donors); + } else { + return (); + } + +} + + + + +=over 4 + +=item find_exon_skipping_events() + +B Finds an exon of gene_1 which reside within an intron of gene_2 + +B gene1, gene2 + +B @skipped_exons + + + notice this is a list of lists + each list is a set of adjacent skipped exons, joined so that they correspond to a single event. +so what we are really getting here is a list of events of skipped exons where each event may contain one or more skipped exons. + + +@skipped_exons = ( + + [ + [exon_lend,exon_rend], ... + + ], + + [ + [exon_lend, exon_rend], ... + + ] + + + ) + +=back + +=cut + + +#### +sub find_exon_skipping_events { + my $self = shift; + my ($gene1, $gene2) = @_; + + # Algorithm: + # -find an internal exon of gene 1 that resides within an intron of gene 2. Flanking exons must be anchored to the other isoform + my %gene1_exons = &enumerate_exons_of_gene($gene1); + my %gene2_introns = &enumerate_introns_of_gene($gene2); + + my @potential_skipped_exons; + foreach my $exon1_lend (keys %gene1_exons) { + my $exon1_rend = $gene1_exons{$exon1_lend}; + + ## See if within intron of second gene + foreach my $intron2_lend (keys %gene2_introns) { + my $intron2_rend = $gene2_introns{$intron2_lend}; + + if ($exon1_lend > $intron2_lend && $exon1_rend < $intron2_rend) { #exon incapsulated in intron + push (@potential_skipped_exons, [$exon1_lend, $exon1_rend]); + } + } + } + + ## Verify flanking exons are anchorable: + my @skipped_exons; + if (@potential_skipped_exons) { + foreach my $potential_skipped_exon (@potential_skipped_exons) { + my ($exon_lend, $exon_rend) = @$potential_skipped_exon; + + ## Try to anchor left exon + my $anchor_left_exon = 0; + foreach my $exon1 ($gene1->get_exons()) { + my ($exon1_lend, $exon1_rend) = sort {$a<=>$b} $exon1->get_coords(); + unless ($exon1_rend < $exon_lend) { next;} + foreach my $exon2 ($gene2->get_exons()) { + my ($exon2_lend, $exon2_rend) = sort {$a<=>$b} $exon2->get_coords(); + unless ($exon2_rend < $exon_lend) { next;} + + if ($exon1_lend < $exon2_rend && $exon1_rend > $exon2_lend) { #anchorable + $anchor_left_exon = 1; + last; + } + } + if ($anchor_left_exon) { last;} + } + unless ($anchor_left_exon) { next;} + + ## Try to anchor the right exon + my $anchor_right_exon = 0; + foreach my $exon1 ($gene1->get_exons()) { + my ($exon1_lend, $exon1_rend) = sort {$a<=>$b} $exon1->get_coords(); + unless ($exon1_lend > $exon_rend) { next;} + foreach my $exon2 ($gene2->get_exons()) { + my ($exon2_lend, $exon2_rend) = sort {$a<=>$b} $exon2->get_coords(); + unless ($exon2_lend > $exon_rend) { next;} + + if ($exon1_lend < $exon2_rend && $exon1_rend > $exon2_lend) { #anchorable + $anchor_right_exon = 1; + last; + } + } + if ($anchor_right_exon) { last;} + } + if ($anchor_right_exon && $anchor_left_exon) { + push (@skipped_exons, $potential_skipped_exon); + } + } + + } + + + ## group into lists of adjacent exons + + my @ret_skipped_exons; + if (@skipped_exons) { + @skipped_exons = sort {$a->[0]<=>$b->[0]} @skipped_exons; + + my %coord_to_order; + ## map each exon to an integer + my $order = 0; + foreach my $exon (sort {$a->{end5}<=>$b->{end5}} $gene1->get_exons()) { + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + $order++; + $coord_to_order{$lend} = $order; + } + + my $first_skipped_exon = shift @skipped_exons; + @ret_skipped_exons = ([$first_skipped_exon]); + + while (@skipped_exons) { + my $last_event = $ret_skipped_exons[$#ret_skipped_exons]; + my $last_skipped_exon = $last_event->[ $#{$last_event} ]; + + my $last_lend = $last_skipped_exon->[0]; + + my $curr_skipped_exon = shift @skipped_exons; + my $curr_lend = $curr_skipped_exon->[0]; + + if ($coord_to_order{$curr_lend} - $coord_to_order{$last_lend} == 1) { + ## adjacent, so group them + push (@$last_event, $curr_skipped_exon); + } + else { + ## not adjacent + # start new event + push (@ret_skipped_exons, [$curr_skipped_exon]); + } + } + } + + return (@ret_skipped_exons); + +} + + + + + + + +=over 4 + +=item find_alternate_exons() + +B Finds terminal exons in gene1 that are different and non-overlapping, and adjacent to overlapping exons. + +B $gene1, $gene2 + +B @range_of_coords_containing_alternate_exons + + @ret = ( { type => lend|rend, + coords => [region_lend,region_rend], + num_exons => intval + } + + , ... + + ) + + + + +=back + +=cut + +sub find_alternate_exons { + my $self = shift; + my ($gene1, $gene2) = @_; + my @alternate_exon_regions; # store coords of alternate exons + # Algorithm: + # -Looking at terminal exons, should have non-overlapping exons prior to the first overlapping exon + + ## Look from front to back: + my @gene1_exons = sort {$a->{end5}<=>$b->{end5}} $gene1->get_exons(); + my @gene2_exons = sort {$a->{end5}<=>$b->{end5}} $gene2->get_exons(); + + my @alternate_exons_front; + for (my $i = 0; $i <= $#gene1_exons; $i++) { + my ($exon1_lend, $exon1_rend) = sort {$a<=>$b} $gene1_exons[$i]->get_coords(); + my $overlapping_j = undef(); + for (my $j = 0; $j <= $#gene2_exons; $j++) { + my ($exon2_lend, $exon2_rend) = sort {$a<=>$b} $gene2_exons[$j]->get_coords(); + + ## check for overlap + if ($exon1_lend < $exon2_rend && $exon1_rend > $exon2_lend) { + $overlapping_j = $j; + last; + } + } + if (defined($overlapping_j)) { + ## See if i and j are not first: + if ($i != 0 && $overlapping_j != 0) { + for (my $x=0; $x < $i; $x++) { + my $exon = $gene1_exons[$x]; + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + push (@alternate_exons_front, [$lend,$rend]); + } + } + last; + } + } + + ## Look from back to front: + + my @alternate_exons_back; + for (my $i = $#gene1_exons; $i >= 0; $i--) { + my ($exon1_lend, $exon1_rend) = sort {$a<=>$b} $gene1_exons[$i]->get_coords(); + my $overlapping_j = undef(); + for (my $j = $#gene2_exons; $j >= 0; $j--) { + my ($exon2_lend, $exon2_rend) = sort {$a<=>$b} $gene2_exons[$j]->get_coords(); + + ## check for overlap + if ($exon1_lend < $exon2_rend && $exon1_rend > $exon2_lend) { + $overlapping_j = $j; + last; + } + } + if (defined($overlapping_j)) { + ## See if i and j are not last: + if ($i != $#gene1_exons && $overlapping_j != $#gene2_exons) { + for (my $x=$#gene1_exons; $x > $i; $x--) { + my $exon = $gene1_exons[$x]; + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + push (@alternate_exons_back, [$lend,$rend]); + } + } + last; + } + } + + if (@alternate_exons_front) { + + my @front_coords; + my $num_alternate_exons_front = scalar (@alternate_exons_front); + foreach my $coordpair (@alternate_exons_front) { + push (@front_coords, @$coordpair); + } + @front_coords = sort {$a<=>$b} @front_coords; + my $region_lend = shift @front_coords; + my $region_rend = pop @front_coords; + push (@alternate_exon_regions, { type => 'lend', + coords => [$region_lend, $region_rend], + num_exons => $num_alternate_exons_front, + } + ); + } + + if (@alternate_exons_back) { + my @back_coords; + my $num_alternate_exons_back = scalar (@alternate_exons_back); + foreach my $coordpair (@alternate_exons_back) { + push (@back_coords, @$coordpair); + } + @back_coords = sort {$a<=>$b} @back_coords; + my $region_lend = shift @back_coords; + my $region_rend = pop @back_coords; + push (@alternate_exon_regions, { type => 'rend', + coords => [$region_lend, $region_rend], + num_exons => $num_alternate_exons_back + } + ); + } + + + return (@alternate_exon_regions); +} + + +=over 4 + +=item find_starts_and_ends_within_introns() + +B The first and last exons of gene_1 are compared to the introns of gene_2. + +B $gene1, $gene2 + +B @coords + +@coords contains the coordinates of either the very end5 or very end3 of terminal exons which fall into introns of gene_2 + +=back + +=cut + + +#### +sub find_starts_and_ends_within_introns { + my $self = shift; + my ($gene1, $gene2) = @_; + + my $fuzzlength = $CDNA::PASA_alignment_assembler::FUZZLENGTH; + + + my %starts_and_ends; + + my $orientation = $gene1->get_orientation(); + + # Algorithm: + # -first and last exon of gene1 is compared to introns of gene2 + my @gene1_exons = $gene1->get_exons(); + my %gene2_introns = &enumerate_introns_of_gene($gene2); + my %gene2_exons = &enumerate_exons_of_gene($gene2); + my $first_exon = $gene1_exons[0]; + my ($end5, $end3) = $first_exon->get_coords(); + + foreach my $intron_lend (keys %gene2_introns) { + my $intron_rend = $gene2_introns{$intron_lend}; + if ($end5 >= $intron_lend && $end5 <= $intron_rend) { #endpoint encapsulated by intron. + + ## make sure it's not fuzz: + if ($orientation eq "+") { + if ( abs ($end5-$intron_rend) + 1 <= $fuzzlength) { + next; + } + } + else { # minus strand + if (abs ($end5-$intron_lend)+1 <= $fuzzlength) { + next; + } + } + + ## make sure exon overlaps another exon + foreach my $exon_lend (keys %gene2_exons) { + my $exon_rend = $gene2_exons{$exon_lend}; + my ($lend, $rend) = sort {$a<=>$b} ($end5, $end3); + if ($rend > $exon_lend && $lend < $exon_rend) { #overlap + $starts_and_ends{start} = $end5; # store start + last; + } + } + last; + } + } + + ## Now try last exon + my $last_exon = $gene1_exons[$#gene1_exons]; + my ($end5, $end3) = $last_exon->get_coords(); + + foreach my $intron_lend (keys %gene2_introns) { + my $intron_rend = $gene2_introns{$intron_lend}; + if ($end3 >= $intron_lend && $end3 <= $intron_rend) { #endpoint encapsulated by intron. + + ## make sure not fuzz: + if ($orientation eq "+") { + if (abs ($end3 - $intron_lend) + 1 <= $fuzzlength) { + next; + } + } + else { #minus strand + if (abs ($end3 - $intron_rend) + 1 <= $fuzzlength) { + next; + } + } + + ## Make sure exon overlaps another exon + foreach my $exon_lend (keys %gene2_exons) { + my $exon_rend = $gene2_exons{$exon_lend}; + my ($lend, $rend) = sort {$a<=>$b} ($end5, $end3); + if ($rend > $exon_lend && $lend < $exon_rend) { + $starts_and_ends{end} = $end3; #store end + last; + } + } + last; + } + } + return (%starts_and_ends); +} + + + +=over 4 + +=item compare_exons() + +B Compares all CDS exons between genes 1 and 2, returns number of identical CDS exons and total number of CDS exons between the two genes. + +B $gene1, $gene2 + +B ($num_identical_CDS_exons, $num_total_CDS_exons) + + +=back + +=cut + + + +#### +sub compare_exons { + my $self = shift; + my ($gene1, $gene2) = @_; + print "gene1_strand: $gene1->{strand}\n"; + my $clone_1 = $gene1->clone_gene(); + print "clone1_strand: " . $clone_1->{strand} . "\n"; + my $clone_2 = $gene2->clone_gene(); + $clone_1->trim_UTRs(); + print "clone1_strand, utrs trimmed: " . $clone_1->{strand} . "\n"; + $clone_2->trim_UTRs(); + + my @exons_1 = $clone_1->get_exons(); + + my @exons_2 = $clone_2->get_exons(); + + my @identity_list = (); + my @all_exons = sort {$a->{end5}<=>$b->{end5}} (@exons_1, @exons_2); + + for (my $i=0; $i <= $#all_exons-1; $i++) { + + my $curr_exon = $all_exons[$i]; + my $next_exon = $all_exons[$i+1]; + + my ($curr_exon_end5, $curr_exon_end3) = $curr_exon->get_coords(); + + my ($next_exon_end5, $next_exon_end3) = $next_exon->get_coords(); + + if ($curr_exon_end5 == $next_exon_end5 && $curr_exon_end3 == $next_exon_end3) { + $identity_list[$i] = 1; + $identity_list[$i+1] = 1; + $i++; #if A = B, then go onto comparing C to D, not B to C. + } + } + + my ($num_identical_exons, $total_num_exons) = (0,0); + for (my $i=0; $i <= $#all_exons; $i++) { + if ($identity_list[$i]) { + $num_identical_exons++; + } + $total_num_exons++; + } + + return ($num_identical_exons, $total_num_exons); +} + + + + +=over 4 + +=item start_or_stop_within_intron() + +B Compares the start codon and stop codon position of gene1 to the introns of gene2. + +B $gene1, $gene2 + +B ($start_within_intron, $stop_within_intron) + +return values are 0|1 meaning true|false for each return parameter. + +=back + +=cut + + +sub start_or_stop_codon_within_intron { + my $self = shift; + my ($gene1, $gene2) = @_; + + + ## Look for annotated start codon or stop codon within intron: + my ($annotated_start_within_intron, $annotated_stop_within_intron) = (0,0); + + my ($start_codon, $stop_codon) = $gene1->get_model_span(); + my (@alignment_segments) = sort {$a->{end5}<=>$b->{end5}} $gene2->get_exons(); + if ($#alignment_segments > 0) { #multiple segments: + for (my $i=1; $i <= $#alignment_segments; $i++) { + my $prev_seg = $alignment_segments[$i-1]; + my ($prev_lend, $prev_rend) = sort {$a<=>$b} $prev_seg->get_coords(); + my $curr_seg = $alignment_segments[$i]; + my ($curr_lend, $curr_rend) = sort {$a<=>$b} $curr_seg->get_coords(); + my ($intron_lend, $intron_rend) = ($prev_rend+1, $curr_lend-1); + if ($start_codon >= $intron_lend && $start_codon <= $intron_rend) { + $annotated_start_within_intron = 1; + } + if ($stop_codon >= $intron_lend && $stop_codon <= $intron_rend) { + $annotated_stop_within_intron = 1; + } + } + } + return ($annotated_start_within_intron, $annotated_stop_within_intron); +} + + + + + + +##### +## Static methods +#### + +=over 4 + +=item adjust_alternate_exon_region_coords() + +B Adjusts the coordinates provided by find_alternate_exons() so that they are extended to include the adjacent intron. + +B gene_obj, region_lend, region_rend + +B adjusted_region_lend, adjusted_region_rend + + +This is useful in cases where we try to see if the variation impacts the protein coding region when compared to an alternate gene. + +region_lend and region_rend are the lend,rend values stored in table: splice_variation under type = 'alternate_exon' + + +=back + +=cut + +sub adjust_alternate_exon_region_coords { + + my ($gene_obj, $region_lend, $region_rend) = @_; + + my @other_exon_coords; + foreach my $exon ($gene_obj->get_exons()) { + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + unless ($lend >= $region_lend && $rend <= $region_rend) { + # not encapsulated: + push (@other_exon_coords, $lend, $rend); + } + } + @other_exon_coords = sort {$a<=>$b} @other_exon_coords; + + #print "other exon coords: @other_exon_coords\n"; + + my $other_lend = shift @other_exon_coords; + my $other_rend = pop @other_exon_coords; + + ## make sure there isn't any overlap + if ($other_lend <= $region_rend && $other_rend >= $region_lend) { + #overlap BAD! + confess ("Error, coordinate regions overlap ($other_lend, $other_rend) w/ region($region_lend, $region_rend)\n" + . Dumper (\@other_exon_coords)); + } + + ## adjust boundary to include intron region + if ($other_rend < $region_lend) { + $region_lend = $other_rend + 1; + } + elsif ($other_lend > $region_rend) { + $region_rend = $other_lend - 1; + } + else { + confess "Error, cannot figure out how to adjust the boundaries." . Dumper (\@other_exon_coords); + } + + return ($region_lend, $region_rend); +} + + +=over 4 + +=item extend_coords_to_intron_bounds() + +B given the coordinates of a region of skipped exons, the coordinates are extended to the far bounds of the adjacent introns + +B (gene_obj, region_lend, region_rend) + +B (adjusted_region_lend, adjusted_region_rend) + +=back + +=cut + + +sub extend_coords_to_intron_bounds { + my ($gene_obj, $region_lend, $region_rend) = @_; + + my @left_coords; + my @right_coords; + + foreach my $exon ($gene_obj->get_exons()) { + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + + if ($lend > $region_rend) { + push (@right_coords, $lend, $rend); + } + elsif ($rend < $region_lend) { + push (@left_coords, $lend, $rend); + } + + } + + @left_coords = sort {$a<=>$b} @left_coords; + @right_coords = sort {$a<=>$b} @right_coords; + + unless (@left_coords && @right_coords) { + confess "Error, missing either left or right coords: \n" + . "left: @left_coords\n" + . "right: @right_coords\n" + . "region: $region_lend, $region_rend\n"; + } + + + my $new_left_bound = pop @left_coords; + $new_left_bound++; + + my $new_right_bound = shift @right_coords; + $new_right_bound--; + + return ($new_left_bound, $new_right_bound); +} + + + + + +1; + + diff --git a/99.scripts/trinity_utils/PerlLib/CDNA/CDNA_alignment.pm b/99.scripts/trinity_utils/PerlLib/CDNA/CDNA_alignment.pm new file mode 100644 index 0000000..9b8498b --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/CDNA/CDNA_alignment.pm @@ -0,0 +1,1874 @@ +#!/usr/local/bin/perl + +########################## +### Class CDNA_alignment +########################## +package main; +our $SEE; + +=head1 NAME + +CDNA::CDNA_alignment + +=cut + +=head1 DESCRIPTION + +This module provides an object specification for storing and manipulating cDNA alignments. The alignment coordinates along the genomic sequence are given along with a genomic sequence. Splice sites are validated and the spliced orientation is determined. + +=cut + + + +package CDNA::CDNA_alignment; +use strict; +use Exons_to_geneobj; +use Gene_obj; +use CDNA::Alignment_segment; +use Carp qw (cluck confess croak); +use GFF_maker; + +## Global Vars +our $ALLOW_ATAC_splice_pairs = 1; # ON by default; turn it off if you prefer. + +## + + +=over 4 + +=item new() + +B Instantiates a new CDNA::CDNA_alignment object. + +B $cdna_length, $Alignment_obj_aref, $sequence_sref + +B<$cdna_length> is the length of the complete cDNA sequence. + +B<$Alignment_obj_aref> is a reference to a list of CDNA::Alignment_segment objects like so: + +$Alignment_obj_aref = \@Alignment_obj_list + +B<$sequence_sref> should be a reference to the string containing the genomic sequence like so: + +$sequence = "gatc....."; + +$sequence_sref = \$sequence; + +B $obj_ref + +returns a reference to a CDNA_alignment object. This object is validated requiring consensus splice sites, and the spliced orientation of the cDNA is determined based on the validating orientation. + +=back + +=cut + + +sub new { + my $packagename = shift; + my ($cdna_length, $Alignment_obj_aref, $sequence_ref) = @_; + unless (@$Alignment_obj_aref) { + die "No alignment segments available to create alignment object.\n"; + } + + my $self = { + acc => undef(), # cDNA accession. + title =>undef(), # com_name, title, header, whatever you want to call the cdna. + genomic_seq => $sequence_ref, + genome_acc => undef, + alignment_segs=>$Alignment_obj_aref, + orientation => undef(), + lend=>undef(), + rend=>undef(), + length=>0, # alignment span length. + num_aligned_nts => 0, #number of cDNA nucleotides found in alignment + cdna_length => $cdna_length, # the length of the cDNA sequence. + avg_per_id => 0, # the average percent identity of this alignment. + error_flag=>0, #default w/o errors. Used to indicate non-consensus splice sites. + spliced_orientation => '?', # [+-?] depending on validating consenus splice sites; relative to genomic sequence orientation. + num_segments=>0, + percent_cdna_aligned => 0, # provides the percentage of the cDNA length that is in the alignment. + is_fli => 0 #indicates whether the cDNA is a full-length insert (ie. expected to be complete). + }; + bless ($self, $packagename); + + ## Convert alignment to object form: + $self->process_alignment(); + + if ($self->{num_segments} > 1 && (ref $sequence_ref)) { + $self->identify_splice_junctions($sequence_ref); + } + return ($self); +} + +sub process_alignment { + my $self = shift; + $self->determine_alignment_attributes(); + my @alignment_segments = $self->get_alignment_segments(); + for (my $i = 0; $i <= $#alignment_segments; $i++) { + my $segment = $alignment_segments[$i]; + if ($#alignment_segments == 0) { #single segment in alignment + $segment->set_type("single"); + } elsif ($i == 0) { + $segment->set_type("first"); + } elsif ($i == $#alignment_segments) { + $segment->set_type("last"); + } else { #must be internal + $segment->set_type("internal"); + } + } + $self->set_num_segments($#alignment_segments + 1); + $self->verify_contiguity(); + +} + +sub identify_splice_junctions { + my $self = shift; + my $sequence_ref = shift; + my $orientation = $self->get_orientation(); + + my $opposite_orientation = ($orientation eq '+') ? '-' : '+'; + + ## Since some cDNAs are supplied in the opposite orientation, the splice sites may be on the reverse strand. + ## In this case, we must change the orientation + my $error_flag = 0; + my $error_text = ""; + my $validating_orientation = $orientation; #initialize. + my %error_lengths; #used to find best orientation (least errors) + foreach my $orient ($orientation, $opposite_orientation) { + $error_text = $self->validate_splice_junctions($orient, $sequence_ref); + $error_lengths{$orient} = length $error_text; + if ($error_text) { + print "ERRORS in Splice sites given orientation $orient:\n$error_text\n" if $::SEE; + unless ($error_flag) { print "Trying other strand might help?\n" if $::SEE;} + $error_text = ""; #rest + $error_flag = 1; + } else { + $error_flag = 0; + $validating_orientation = $orient; + last; + } + } + if ($error_flag) { + print "Sorry, still contains problematic splice sites:\n$error_text\n" if $::SEE; + print "Setting error flag for this alignment\n" if $::SEE; + $self->set_error_flag("Splice site validations failed"); + print "ERROR: SETTING ERROR_FLAG\n" if $::SEE; + ## revalidate splice sites using orientation that generated the least errors: + my @orients = sort {$error_lengths{$a}<=>$error_lengths{$b}} ('+', '-'); + my $best_orient = shift @orients; + $self->validate_splice_junctions($best_orient, $sequence_ref); #required for appropriate token printing. + } else { + $self->set_spliced_orientation($validating_orientation); + } +} + +sub validate_splice_junctions { + my $self = shift; + my ($orient) = shift; + my $sequence_ref = shift; + my $errors = ""; + my (@splice_boundary_pairs) = &get_consensus_splice_sites($orient); + my @segments = $self->get_alignment_segments(); + + my $num_segments = scalar (@segments); + if ($num_segments == 1) { + ## no introns + die "Error, trying to validate splice junctions for single segment alignment! "; + } + + ## analyze introns + for (my $i = 1; $i <= $#segments; $i++) { + my ($prev_segment, $curr_segment) = ($segments[$i-1], $segments[$i]); + + my ($prev_lend, $prev_rend) = $prev_segment->get_coords(); + my ($curr_lend, $curr_rend) = $curr_segment->get_coords(); + + my $splice_chars_left = uc substr($$sequence_ref, $prev_rend, 2); + my $splice_chars_right = uc substr($$sequence_ref, $curr_lend -2 -1, 2); + + $prev_segment->set_right_splice_site_chars($splice_chars_left); + $curr_segment->set_left_splice_site_chars($splice_chars_right); + + ## check boundaries: + my $splice_pair_OK = 0; + + CONSENSUS_PAIR: + foreach my $consensus_pair (@splice_boundary_pairs) { + my ($left_chars, $right_chars) = @$consensus_pair; + + if ($left_chars eq $splice_chars_left && $right_chars eq $splice_chars_right) { + # found consensus + ## further validate any AT-AC for donor site extended consensus + my ($intron_lend, $intron_rend) = ($prev_rend + 1, $curr_lend - 1); + if ($ALLOW_ATAC_splice_pairs) { + if ( ( $orient eq '+' && $left_chars eq 'AT') + || + ($orient eq '-' && $right_chars eq 'AT') ) { + unless (&_validates_AT_AC_donor_extended_consensus($orient, $intron_lend, $intron_rend, $sequence_ref)) { + next CONSENSUS_PAIR; + } + } + } + + ## got consensus splice pair + $splice_pair_OK = 1; + ## left and right local are relative to intron + # in methods, they reference segment + $prev_segment->set_right_splice_junction(1); + $curr_segment->set_left_splice_junction(1); + last; + } + } + unless ($splice_pair_OK) { + + $errors .= "nonconsensus splice pair [$splice_chars_left-$splice_chars_right]\n"; + + ## validate boundaries separately. + ## this is useful if only one site is nonconsensus + foreach my $consensus_pair (@splice_boundary_pairs) { + my ($left_chars, $right_chars) = @$consensus_pair; + if ($left_chars eq $splice_chars_left) { + $prev_segment->set_right_splice_junction(1); + } + if ($right_chars eq $splice_chars_right) { + $curr_segment->set_left_splice_junction(1); + } + } + + ## make nonconsensus lower case + unless ($prev_segment->has_right_splice_junction()) { + $prev_segment->set_right_splice_site_chars( lc $splice_chars_left); + } + unless ($curr_segment->has_left_splice_junction()) { + $curr_segment->set_left_splice_site_chars( lc $splice_chars_right); + } + + } + } + + return ($errors); +} + + +sub verify_contiguity { + my $self = shift; + my $num_segments = $self->get_num_segments(); + my $orient = $self->get_orientation(); + if ($num_segments > 1) { + my @alignment_segments = $self->get_alignment_segments(); + for (my $i = 1; $i <= $#alignment_segments; $i++) { + my $prev_seg = $alignment_segments[$i-1]; + my $curr_seg = $alignment_segments[$i]; + my $error_flag = 0; + my $diff = $curr_seg->{mlend} - $prev_seg->{mrend}; + if ($orient eq '+' && $diff != 1) { + $error_flag = 1; + }elsif ($orient eq '-' && $diff != -1) { + $error_flag = 1; + } + if ($error_flag) { + $self->set_error_flag("Incontiguous alignment"); + return(); + } + } + } +} + + + +sub get_consensus_splice_sites () { + + my $orientation = shift; + + my @pairs; + + if ($orientation eq '+') { + + ## Forward pairs + # GT-AG + # GC-AG + # AT-AC + + push (@pairs, ['GT', 'AG'], ['GC', 'AG']); + + if ($ALLOW_ATAC_splice_pairs) { + push (@pairs, ['AT', 'AC']); + } + + } + + else { + ## Rev Comp of above: + # CT-AC + # CT-GC + # GT-AT + + push (@pairs, ['CT', 'AC'], ['CT', 'GC']); + + if ($ALLOW_ATAC_splice_pairs) { + push (@pairs, ['GT', 'AT']); + } + } + + + return (@pairs); +} + + +#### +sub _validates_AT_AC_donor_extended_consensus { + my ($orient, $intron_lend, $intron_rend, $sequence_ref) = @_; + + my $forward_consensus = 'ATATCC'; + my $reverse_consensus = 'GGATAT'; + + if ($orient eq '+') { + my $long_donor_seq = uc substr($$sequence_ref, $intron_lend - 1, 6); + if ($long_donor_seq eq $forward_consensus) { + return (1); + } + } + elsif ($orient eq '-') { + my $long_donor_seq = uc substr($$sequence_ref, $intron_rend -6, 6); + if ($long_donor_seq eq $reverse_consensus) { + return (1); + } + } + + ## got here, didn't fit long consensus + return (0); +} + + + + + +sub set_orientation { + my $self = shift; + my $orientation = shift; + $self->{orientation} = $orientation; +} + + +=over 4 + +=item get_aligned_orientation() + +B Provides the orientation of the incoming cDNA alignment. + +B none. + +B [+-] + +=back + +=cut + + +sub get_aligned_orientation { + my $self = shift; + return ($self->{orientation}); +} + +sub get_orientation { # deprecated in favor of get_aligned_orientation() + my $self = shift; + return ($self->get_aligned_orientation()); +} + + +sub set_title { + my $self = shift; + my $title = shift; + $self->{title} = $title; +} + + + + +=over 4 + +=item get_title() + +B Provides the title for the cDNA, generally the header from a fasta file + +B none. + +B string or undef + +=back + +=cut + + + + +sub get_title { + my $self = shift; + return ($self->{title}); +} + + + +sub set_coords { + my $self = shift; + my ($lend, $rend) = @_; + $self->{lend} = $lend; + $self->{rend} = $rend; +} + + +=over 4 + +=item get_coords() + +B Provides the coordinate span for the alignment along the genomic sequence. Orientation is not implied, coordinates always provided relevant to the forward orientation. Use the get_orientation method to determine the strand. + +B none. + +B $lend, $rend + +$lend, $rend are integer values indicating the beginning and end coordinates of the alignment on the genomic sequence. The genomic sequence begins at position 1. + +=back + +=cut + + + +sub get_coords { + my $self = shift; + return ($self->{lend}, $self->{rend}); +} + + + + + +=over 4 + +=item get_mcoords() + +B returns the cDNA sequence coordinates corresponding to the lend and rend returned by get_coords() + + ie. my ($lend, $rend) = $alignment->get_coords(); // always in forward orientation (lend < rend) + + my ($mlend, $mrend) = $alignment->get_mcoords(); + + $mlend corresponds to $lend + + $mrend corresponds to $rend + + +Bnone + +B ($mlend, $mrend) + +=back + +=cut + + +sub get_mcoords { + my $self = shift; + + my @alignment_segments = $self->get_alignment_segments(); + + my $leftmost_seg = shift @alignment_segments; + my $rightmost_seg = pop @alignment_segments; + + unless ($rightmost_seg) { + $rightmost_seg = $leftmost_seg; #must only be one in which case left = right + } + + my ($mlend, $whatever1) = $leftmost_seg->get_mcoords(); + my ($whatever2, $mrend) = $rightmost_seg->get_mcoords(); + + return ($mlend, $mrend); +} + + + + +=over 4 + +=item get_intron_coords() + +B returns list of intron coordinates + +B none + +B ( [intron_lend, intron_rend], [intron_lend, intron_rend], ...) + + + lend always less than rend + +=back + +=cut + + +sub get_intron_coords { + my $self = shift; + + my @segments = $self->get_alignment_segments(); + + my @seg_coords; + foreach my $segment (@segments) { + my ($exon_lend, $exon_rend) = sort {$a<=>$b} $segment->get_coords(); + + push (@seg_coords, [$exon_lend, $exon_rend]); + } + + my @intron_coords; + + if (scalar (@seg_coords) >= 2) { + ## actually have introns + @seg_coords = sort {$a->[0]<=>$b->[0]} @seg_coords; + + my $curr_seg = shift @seg_coords; + while (@seg_coords) { + my $next_seg = shift @seg_coords; + + my ($curr_lend, $curr_rend) = @$curr_seg; + + my ($next_lend, $next_rend) = @$next_seg; + + my ($intron_lend, $intron_rend) = ($curr_rend + 1, $next_lend - 1); + + push (@intron_coords, [$intron_lend, $intron_rend]); + + $curr_seg = $next_seg; + } + } + + + return (@intron_coords); +} + + +sub determine_alignment_attributes { + my $self = shift; + my @alignment_segments = $self->get_alignment_segments(); + my @coords; + my $orientation; + my $num_nts_matched = 0; + my $per_id_x_length = 0; + foreach my $segment (@alignment_segments) { + my ($lend, $rend) = $segment->get_coords(); + my ($mlend, $mrend) = $segment->get_mcoords(); + my $orient = $segment->get_orientation(); + if (!$orientation && $orient =~ /[+-]/) { + $orientation = $orient; + } + my $seg_length; + if ($mrend && $mlend) { + $seg_length = abs($mrend - $mlend) + 1; + } else { + $seg_length = abs ($rend - $lend) + 1; + } + $num_nts_matched += $seg_length; + my $per_id = $segment->get_per_id(); + $per_id_x_length += $per_id * $seg_length; + push (@coords, $lend, $rend); + } + $self->set_orientation($orientation); + foreach my $segment (@alignment_segments) { + $segment->set_orientation($orientation); + } + @coords = sort {$a<=>$b} @coords; + my $lend = shift @coords; + my $rend = pop @coords; + $self->set_coords($lend, $rend); + $self->{length} = abs ($rend - $lend) + 1; + $self->{num_aligned_nts} = $num_nts_matched; + $self->{avg_per_id} = $per_id_x_length / $num_nts_matched; + if ($self->{avg_per_id} > 100) { die "Error, can't have average per_id > 100%\n";} + if ($self->{cdna_length} > 0) { + $self->{percent_cdna_aligned} = $num_nts_matched / $self->{cdna_length} * 100; + } +} + + +=over 4 + +=item get_alignment_segments() + +B Returns the alignment segments which comprise an alignment object, ordered by left coordinate position. + +B none + +B @alignment_segments + +@alignment_segments is a list of CDNA::Alignment_segment objects (see below). + +=back + +=cut + +sub get_alignment_segments() { + my $self = shift; + return (sort {$a->{lend}<=>$b->{lend}} @{$self->{alignment_segs}}); +} + + + +=over 4 + +=item add_alignment_segment() + +B Adds a CDNA::Alignment_segment object to the list of segments of this CDNA_alignment object. + +B CDNA::Alignment_segment object. + +B none. + +=back + +=cut + + +sub add_alignment_segment() { + my $self = shift; + my $segment = shift; + push (@{$self->{alignment_segs}}, $segment); +} + + +=over 4 + +=item delete_all_segments() + +B Empties the current list of Alignment_segment objects. + +B None. + +B None. + +=back + +=cut + + + + +sub delete_all_segments () { + my $self = shift; + @{$self->{alignment_segs}} = (); #empty the array +} + + +sub set_error_flag () { + my $self = shift; + my $error = shift; + if ($self->{error_flag}) { + $self->{error_flag} .= $error; + } else { + $self->{error_flag} = $error; + } +} + + +=over 4 + +=item get_error_flag() + +B Provides the status of the aligments validation. + +B none. + +B [error_text|0] + +A text string containing the error is returned if an error exists. Otherwise, zero is returned indicating the lack of errors. + +=back + +=cut + + +sub get_error_flag () { + my $self = shift; + return ($self->{error_flag}); +} + + +=over 4 + +=item toString() + +B Returns an alignment as lines of text providing coordinate information. + +B none. + +B alignment_string + +=back + +=cut + + +sub toString() { + my $self = shift; + my $text = "\n\nAlignment: orientation: " . $self->{orientation} . "\n" + . "coords: " . $self->{lend} . "-" . $self->{rend} . "\n"; + + my @segments = $self->get_alignment_segments(); + foreach my $segment (@segments) { + $text .= $segment->toString(); + } + return ($text); +} + + +#### +sub to_GFF3_format { + my $self = shift; + my %preferences = @_; + + + my $seq_id = $preferences{seq_id} || $self->{genome_acc} || confess "Need seq_id in preferences, or set genome_acc attribute of obj"; + my $match_id = $preferences{match_id} or confess "Error, require match_id attribute"; + my $source = $preferences{source} || "PASA"; + + my $orientation = $self->get_orientation(); + + + + my @alignment_segments = $self->get_alignment_segments(); + my $acc = $self->get_acc() or confess "Error, accession for cdna_alignment is not available"; + + + my $GFF3_text = ""; + + foreach my $alignment_segment (@alignment_segments) { + my ($genome_lend, $genome_rend) = $alignment_segment->get_coords(); + my ($cdna_lend, $cdna_rend) = sort {$a<=>$b} $alignment_segment->get_mcoords(); + + + my $gff_struct = { seq_id => $seq_id, + source => $source, + type => "cDNA_match", + lend => $genome_lend, + rend => $genome_rend, + strand => $orientation, + attributes => "ID=$match_id; Target=$acc $cdna_lend $cdna_rend +", + }; + + if (my $per_id = $alignment_segment->get_per_id()) { + $gff_struct->{score} = $per_id; + } + + $GFF3_text .= &GFF_maker::get_GFF_line($gff_struct); + + } + + return ($GFF3_text); + +} + + + + +#### +sub to_GTF_format { + my $self = shift; + my %preferences = @_; + + + my $seq_id = $preferences{seq_id} || $self->{genome_acc} || confess "Need seq_id in preferences, or set genome_acc attribute of obj"; + + my $gene_id = $preferences{gene_id} || confess "Need gene_id in preferences"; + my $transcript_id = $preferences{transcript_id} || $self->get_acc(); + + + my $source = $preferences{source} || "PASA"; + + my $orientation = $self->get_orientation(); + + + my ($trans_lend, $trans_rend) = sort {$a<=>$b} $self->get_coords(); + + my @alignment_segments = $self->get_alignment_segments(); + my $acc = $self->get_acc() or confess "Error, accession for cdna_alignment is not available"; + + + my $gtf_text = join("\t", ( $seq_id, + $source, + "transcript", + $trans_lend, + $trans_rend, + ".", + $orientation, + ".", + "gene_id \"$gene_id\"; transcript_id \"$transcript_id\";") + ) . "\n"; + + + foreach my $alignment_segment (@alignment_segments) { + my ($genome_lend, $genome_rend) = $alignment_segment->get_coords(); + my ($cdna_lend, $cdna_rend) = sort {$a<=>$b} $alignment_segment->get_mcoords(); + + + my $per_id = $alignment_segment->get_per_id() || "."; + + $gtf_text .= join("\t", ( $seq_id, + $source, + "exon", + $genome_lend, + $genome_rend, + $per_id, + $orientation, + ".", + "gene_id \"$gene_id\"; transcript_id \"$transcript_id\";") + ) . "\n"; + + + + } + + return ($gtf_text); + +} + + +sub provide_cdna_segment_coords () { + my $self = shift; + my @segments = $self->get_alignment_segments(); + my $coord_mapper = $self->{genome_cdna_coord_mapper}; + foreach my $segment (@segments) { + my ($lend, $rend) = $segment->get_coords(); + my $mlend = $coord_mapper->{$lend}; + my $mrend = $coord_mapper->{$rend}; + $segment->set_mcoords($mlend, $mrend); + } +} + +=over 4 + +=item remap_cdna_segment_coords() + +B Method is used on an assembled alignment to renumber the coordinates of the assembled cDNA product, starting at 1 and ending at the assembly length. Orientation is determined by the validated spliced orientation. + +B none. + +B none. + +=back + +=cut + +sub remap_cdna_segment_coords () { ## Used when cDNA mapped coordinates aren't known, or when an assembly of other alignments was generated. + ## reassigns cdna coords to alignment coords starting at 1 and ending at determined length. + ## spliced orientation determines how the coords will map + ## if spliced orient is ambiguous, aligned orient is used. + + my $self = shift; + my %coord_mapper; + + + my @alignment_segments = $self->get_alignment_segments(); #remember, already in increasing order. + + my $spliced_orient = $self->get_spliced_orientation(); + my $aligned_orient = $self->get_aligned_orientation(); + + my $transcript_orientation = ($spliced_orient =~ /[\+\-]/) ? $spliced_orient : $aligned_orient; + + if ($transcript_orientation eq '-') { + @alignment_segments = reverse @alignment_segments; + } + + my $curr_pos = 0; + foreach my $segment (@alignment_segments) { + my ($lend, $rend) = $segment->get_coords(); + my $seglength = abs ($rend - $lend) + 1; + my $mlend = $curr_pos + 1; + my $mrend = $curr_pos + $seglength; + $curr_pos += $seglength; + if ($transcript_orientation eq '-') { + ($mlend, $mrend) = ($mrend, $mlend); + } + + $segment->set_orientation($transcript_orientation); + #print "setting $mlend, $mrend\n"; + $coord_mapper{$lend} = $mlend; + $coord_mapper{$rend} = $mrend; + $segment->set_mcoords($mlend, $mrend); + } + + ## reset aligned_orient to transcript_orient if different + ## this should only happen when aligned orient and spliced orient are opposite. + ## in which case, the aligned orient is reset to the spliced orient. + + if ($aligned_orient ne $transcript_orientation) { + $self->set_orientation($transcript_orientation); + } + +} + +=over 4 + +=item toToken() + +B similar to the toString() method, but returns a pretty line of text summarizing the alignment and splice site data. + +B none. + +B text_line + +Here is an example: + +orient(-/-) align: 103762(2613)-104359(2016)ECT....ACE104452(2015)-105229(1238)ECT....ACE105315(1237)-105482(1070)ECT....ACE105582(1069)-105842(809)ECT....ACE105935(808)-105985(758)ECT....ACE106071(757)-106316(512)ECT....ACE106394(511)-106619(286)ECT....ACE106712(285)-106879(118)ECT....ACE107427(117)-107543(1) + +or + +orient(+/+) align: 83074(1)-83318(245)EGT....AGE83637(246)-83702(311)EGT....AGE83796(312)-83846(362)EGT....AGE83938(363)-84017(442)EGT....AGE84308(443)-84352(487)EGT....AGE84467(488)-84507(528) + + +The orient specification includes (cDNA sequence alignment orientation/ spliced orientation). These are different when the reverse-complement of the sequence is provided, ascertained by the aligned orientation with consensus splice sites. + +=back + +=cut + +sub toToken () { + my $self = shift; + my $orientation = $self->get_orientation(); + my $spliced_orientation = $self->get_spliced_orientation(); + my @alignment_segments = $self->get_alignment_segments(); + my $assembled_token = "orient(a$orientation/s$spliced_orientation) align: "; + for (my $i = 0; $i <= $#alignment_segments; $i++) { + my $segment = $alignment_segments[$i]; + $assembled_token .= $segment->toToken(); + unless ($i == $#alignment_segments) { + $assembled_token .= "...."; + } + } + return ($assembled_token); +} + + +sub set_num_segments { + my $self = shift; + my $num_segments = shift; + $self->{num_segments} = $num_segments; +} + +=over 4 + +=item get_num_segments() + +B method provides the number of segments composing an alignment. + +B none. + +B int + +=back + +=cut + +sub get_num_segments { + my $self = shift; + return ($self->{num_segments}); +} + +sub set_spliced_orientation { + my $self = shift; + my $orientation = shift; + $self->{spliced_orientation} = $orientation; +} + + +=over 4 + +=item get_spliced_orientation () + +B provides the validating spliced orientation for an alignment. In some cases this will be different from the alignment orientation; for example, in cases where the cDNA sequence is provided in the reverse orientation. + +B none + +B [+|-|undef()] + +undef is returned if the alignment did not validate properly. See get_error_flag() + +=back + +=cut + + +sub get_spliced_orientation { + my $self = shift; + return ($self->{spliced_orientation}); +} + + +=over 4 + +=item set_acc() + +B Method sets the accession field of the cDNA sequence. + +B string + +Provide the accession for the cDNA corresponding to this alignment. + +B none. + +=back + +=cut + +sub set_acc () { + my $self = shift; + my $acc = shift; + $self->{acc} = $acc; +} + +=over 4 + +=item get_acc() + +B Method provides the accession for the cDNA in the alignment. + +B none + +B string + +=back + +=cut + + +sub get_acc () { + my $self = shift; + return ($self->{acc}); +} + + +=over 4 + +=item set_fli_status() + +B sets the is_fli attribute of the cDNA alignment, indicative of a full-length insert clone (or complete cDNA sequence). + +B [1|0] + +1 = true, 0 = false. + +B [1|0] + +=back + +=cut + + +sub set_fli_status { + my $self = shift; + my $is_fli_status = shift; + $self->{is_fli} = $is_fli_status; +} + + +=over 4 + +=item is_fli() + +BProvides the full-length insert status of the cDNA. + +B none. + +B [0|1] + +=back + +=cut + +sub is_fli { + my $self = shift; + return ($self->{is_fli}); +} + + + + + + +=over 4 + +=item toAlignIllustration() + +B Provides a single line of text which illustrates the gapped alignment. See the example below. + +B none. + +B string + +Here is an example of an illustrated alignment: + +------> <---> <----- (-)asmbl_6711 + +=back + +=cut + + +sub toAlignIllustration () { + my ($self, $subtract, $rel_max, $max_line_chars) = @_; + my $spliced_orient = $self->get_spliced_orientation(); + my $orient = $self->get_orientation(); + my @segments = $self->get_alignment_segments(); + my @chars = (); + my $converter = sub {my $coord = shift; + return ( int ( ($coord - $subtract)/$rel_max * $max_line_chars + 0.5)); + }; + foreach my $segment (@segments) { + my ($lend, $rend) = $segment->get_coords(); + my $l_rel = &$converter($lend); + #print "lend: $lend -> l_rel: $l_rel\n" if $::SEE; + my $r_rel = &$converter($rend); + #print "rend: $rend -> r_rel: $r_rel\n" if $::SEE; + + for (my $i = $l_rel; $i <= $r_rel; $i++) { + $chars[$i] = '-'; + } + if ($segment->has_left_splice_junction()) { + $chars[$l_rel] = '<'; + } elsif ( (! $segment->is_first()) && (! $segment->is_single_segment())) { + $chars[$l_rel] = '|'; + } + + if ($segment->has_right_splice_junction()) { + $chars[$r_rel] = '>'; + } elsif ( (! $segment->is_last()) && (! $segment->is_single_segment())) { + $chars[$r_rel] = '|'; + } + } + + #fill rest of line with spaces. + for (my $i = 0; $i <= $#chars; $i++) { + unless ($chars[$i]) { + $chars[$i] = ' '; + } + } + my $outline = join ("", @chars); + my $acc = $self->get_acc(); + $acc =~ tr/\t\n\000-\037\177-\377/\t\n/d; #remove any control characters from accession. + my $fli_status = ($self->is_fli()) ? " FL" : "";; + return ($outline . "\t(a$orient/s$spliced_orient)" . $acc . $fli_status); +} + + +=over 4 + +=item get_gene_obj_via_alignment() + +B Creates a Gene_obj object based on an alignment using the Exons_to_geneobj.pm module. + +B + +B Gene_obj + +The object returned is of the type Gene_obj defined in Gene_obj.pm + +=back + +=cut + + +sub get_gene_obj_via_alignment { + my $self = shift; + my $partial_info_href = shift; # { 5prime => 0|1, 3prime => 0|1 }, optional + + my $orient = $self->get_spliced_orientation(); + + if ($orient eq '+' || '-') { + return($self->_get_gene_obj_via_alignment_by_orient($orient, $partial_info_href)); + } + elsif ($orient eq '?') { + ## find the orientation that provides the longest ORF + + my $plus_orient_gene = $self->_get_gene_obj_via_alignment_by_orient('+', $partial_info_href); + my $minus_orient_gene = $self->_get_gene_obj_via_alignment_by_orient('-', $partial_info_href); + + if ($plus_orient_gene->get_CDS_length() >= $minus_orient_gene->get_CDS_length()) { + return($plus_orient_gene); + } + else { + return($minus_orient_gene); + } + + } + else { + confess "cannot process spliced orientation of $orient "; + } +} + + + + + + +sub _get_gene_obj_via_alignment_by_orient { + my ($self, $orient, $partial_info_href) = @_; + + my @alignment_segments = $self->get_alignment_segments(); + my %coords; + foreach my $segment (@alignment_segments) { + my ($end5, $end3) = $segment->get_coords(); + + if ($orient eq '-') { + ($end5, $end3) = ($end3, $end5); #force coordinates to contain orientation info. + } + $coords{$end5} = $end3; + } + my $genomic_seq_ref = $self->{genomic_seq}; + my $gene_obj; + if ($genomic_seq_ref) { + ## find ORF + $gene_obj = Exons_to_geneobj::create_gene_obj(\%coords, $genomic_seq_ref, $partial_info_href); + } else { + ## No ORF + $gene_obj = new Gene_obj; + $gene_obj->populate_gene_obj({}, \%coords); + } + + return ($gene_obj); +} + + + + + +=over 4 + +=item force_spliced_validation() + +B Routine used for testing purposes. Any fake alignment can be created and set to validate to the corresponding orientation using this routine. + +B [+|-] + +B none. + +=back + +=cut + + +sub force_spliced_validation { + my $self = shift; + my $orientation = shift; + + unless ($orientation eq '+' || $orientation eq '-') { + croak ("cannot force spliced validation to $orientation.\n"); + return; + } + + my ($right_splice, $left_splice) = ($orientation eq '+') ? ('XX','YY') : ('YY','XX'); + + my @alignment_segments = $self->get_alignment_segments(); + foreach my $alignment_segment (@alignment_segments) { + if ($alignment_segment->is_internal() || $alignment_segment->is_last()) { + $alignment_segment->set_left_splice_junction(1); + $alignment_segment->set_left_splice_site_chars($left_splice); + } + if ($alignment_segment->is_internal() || $alignment_segment->is_first()) { + $alignment_segment->set_right_splice_junction(1); + $alignment_segment->set_right_splice_site_chars($right_splice); + } + $alignment_segment->set_orientation($orientation); + } + + $self->set_spliced_orientation($orientation); +} + + + +=over 4 + +=item clone() + +BClones a CDNA_alignment object into a new CDNA_alignment object with same attributes. Performs a deep copy, so all alignment segments contained within the cloned CDNA_alignment object are also clones. + +B none. + +B new CDNA_alignment + +=back + +=cut + +sub clone { + my $self = shift; + my $packagename = ref $self; + + my $clone = {}; + bless ($clone, $packagename); + foreach my $key (keys %$self) { + $clone->{$key} = $self->{$key}; + } + $clone->{alignment_segs} = []; + foreach my $alignment_segment ($self->get_alignment_segments()) { + $clone->add_alignment_segment($alignment_segment->clone()); + } + + return ($clone); +} + + + + +=over 4 + +=item get_genomic_seq_ref() + +B Returns a scalar refernence to the genomic sequence. + +B none. + +B string_ref + +=back + +=cut + + + +sub get_genomic_seq_ref { + my $self = shift; + return ($self->{genomic_seq}); +} + + +=over 4 + +=item extractSplicedSequence() + +B Returns a string corresponding to the spliced cDNA sequence. + +B none. + +B scalar + +=back + +=cut + + + +sub extractSplicedSequence { + my $self = shift; + my ($genomic_seq_ref) = @_; + my $genomic_seq = $genomic_seq_ref; + + unless ($genomic_seq) { + $genomic_seq = $self->{genomic_seq}; + unless (ref $genomic_seq) { + confess "Can't extract the spliced sequence when no genomic sequence reference is available.\n"; + } + } + + my @segments = $self->get_alignment_segments(); + my $splicedSequence = ""; + my $toggle = 0; + foreach my $segment (@segments) { + my ($lend, $rend) = $segment->get_coords(); + my $length = abs ($rend - $lend) + 1; + my $exonseq = substr ($$genomic_seq, $lend - 1, $length); + + # alternate case among segments to facilitate manual identification of junctions. + if ($toggle) { + $exonseq = lc $exonseq; + $toggle = 0; + } + else { + $exonseq = uc $exonseq; + $toggle = 1; + } + + $splicedSequence .= $exonseq; + } + my $orient = $self->get_orientation(); + if ($orient eq "-") { + #reverse complement the sequence: + $splicedSequence = reverse ($splicedSequence); + $splicedSequence =~tr/ACGTacgtyrkmYRKM/TGCAtgcarymkRYMK/; + } + return ($splicedSequence); +} + + + + + +=over 4 + +=item get_cDNA_to_genomic_coordinates() + +B converts a cDNA sequence -relative coordinates to the corresponding genomic sequence coordinates. + +B @coordinates + +list of integers + +B @converted_coordinates + +list of integers. + +=back + +=cut + +sub get_cDNA_to_genomic_coordinates { + my $self = shift; + my @coordinates = @_; + + my @ret_coordinates = (); + + my @segments = $self->get_alignment_segments(); + foreach my $cdna_coord (@coordinates) { + my $corresponding_segment; + foreach my $seg (@segments) { + my ($mlend, $mrend) = sort {$a<=>$b} $seg->get_mcoords(); + if ($cdna_coord >= $mlend && $cdna_coord <= $mrend) { + $corresponding_segment = $seg; + last; + } + } + unless (ref $corresponding_segment) { + confess "Error, cDNA coordinate ($cdna_coord) not found in segment list: " . $self->toToken(); + } + my ($lend, $rend) = $corresponding_segment->get_coords(); + my ($mlend, $mrend) = $corresponding_segment->get_mcoords(); + my $orient = $corresponding_segment->get_orientation(); + + my $genomic_coord = undef; + if ($orient eq '+') { + my $diff = $cdna_coord - $mlend; + $genomic_coord = $lend + $diff; + } elsif ($orient eq '-') { + ## mlend > mrend + my $diff = $cdna_coord - $mrend; + $genomic_coord = $rend - $diff; + } + + push (@ret_coordinates, $genomic_coord); + } + + return (@ret_coordinates); +} + + + + +=item get_genomic_to_cDNA_coordinates() + +B converts coordinates in the genome to coordinates in the cDNA: + +B @coordinates + +list of integers + +B @converted_coordinates + +list of integers. Undef is returned for each position that could not be converted to the genomic coordinate system because it was not found within the aligned region. + +=back + +=cut + + +#### +sub get_genomic_to_cDNA_coordinates { + + my $self = shift; + my @coordinates = @_; + + my @ret_coordinates = (); + + my @segments = $self->get_alignment_segments(); + foreach my $genomic_coord (@coordinates) { + my $corresponding_segment; + foreach my $seg (@segments) { + my ($lend, $rend) = $seg->get_coords(); + if ($genomic_coord >= $lend && $genomic_coord <= $rend) { + $corresponding_segment = $seg; + last; + } + } + unless (ref $corresponding_segment) { + confess "Error, genomic coordinate ($genomic_coord) not found in segment list:" . $self->toToken(); + } + my ($lend, $rend) = $corresponding_segment->get_coords(); + my ($mlend, $mrend) = $corresponding_segment->get_mcoords(); + my $orient = $corresponding_segment->get_orientation(); + my $diff = $rend - $genomic_coord; + my $cdna_coord = undef; + if ($orient eq '+') { + $cdna_coord = $mrend - $diff; + } elsif ($orient eq '-') { + $cdna_coord = $mrend + $diff; + } + + push (@ret_coordinates, $cdna_coord); + } + + return (@ret_coordinates); +} + + + +#### +sub overlaps_genome_span { + my ($self, $other_alignment) = @_; + + my ($lend_A, $rend_A) = sort {$a<=>$b} $self->get_coords(); + + my ($lend_B, $rend_B) = sort {$a<=>$b} $other_alignment->get_coords(); + + if (&_overlap($lend_A, $rend_A, $lend_B, $rend_B)) { + return(1); + } + else { + return(0); + } +} + + +sub has_overlapping_segment { + my ($self, $other_alignment) = @_; + + unless ($self->overlaps_genome_span($other_alignment)) { + return(0); + } + + + my @self_segments = $self->get_alignment_segments(); + + my @other_segments = $self->get_alignment_segments(); + + foreach my $segment_A (@self_segments) { + + my ($lend_A, $rend_A) = sort {$a<=>$b} $segment_A->get_coords(); + + + foreach my $segment_B (@other_segments) { + + my ($lend_B, $rend_B) = sort {$a<=>$b} $segment_B->get_coords(); + + if (&_overlap($lend_A, $rend_A, $lend_B, $rend_B) ) { + + return(1); + } + + } + + } + + return(0); # no overlapping segment +} + + +=over 4 + +=item is_compatible() + +B Returns true (1) if this alignment is found compatible with the other_alignment_obj + +B ($other_alignment_obj, $fuzz_dist) + +Alignments A and B are of type CDNA::CDNA_alignment + +B [1|0] + +Compatibility between alignment objects requires: +-within their region of overlap, introns are identical +-if both have spliced orientations, they must be on the same strand + + +=back + +=cut + + +#### +sub is_compatible { + my $self = shift; + my ($other_alignment, $fuzzlength) = @_; + if (!defined $fuzzlength) { + $fuzzlength = 0; ## no fuzzy termini allowed + } + + ## The compatibility test requires: + # -alignments must have the same spliced orientation if not ? + # -alignments must overlap + # -alignments must have identical introns in their region of overlap (taking into account the fuzz distance for terminal exons) + + my ($a_lend, $a_rend) = $self->get_coords(); + my $a_spliced_orient = $self->get_spliced_orientation(); + my $a_num_segments = $self->get_num_segments(); + + my ($b_lend, $b_rend) = $other_alignment->get_coords(); + my $b_spliced_orient = $other_alignment->get_spliced_orientation(); + my $b_num_segments = $other_alignment->get_num_segments(); + + ## overlap test: + unless (&_overlap($a_lend, $a_rend, $b_lend, $b_rend)) { + return (0); # not compatible + } + + ## transcribed orientation test: + if ($a_spliced_orient ne $b_spliced_orient && $a_spliced_orient ne '?' && $b_spliced_orient ne '?') { + # neither alignment is ambiguously oriented and they're transcribed on opposite strands: + return (0); # not compatible + } + + ## same introns test: + if ($a_num_segments > 1 || $b_num_segments > 1) { + my @a_introns = $self->get_intron_coords(); + my @b_introns = $self->get_intron_coords(); + + my ($overlapping_lend, $overlapping_rend) = &_get_coords_of_overlap($a_lend, $a_rend, $b_lend, $b_rend); + print "Overlapping coords between alignments: $overlapping_lend to $overlapping_rend\n" if $SEE; + ## make adjustments to required overlap coordinates considering fuzzlength: + if ($fuzzlength) { + ## adjust left overlap boundary requirement + $overlapping_lend = &_adjust_left_overlap_boundary_via_fuzzlength($overlapping_lend, $self, $other_alignment, $fuzzlength); + $overlapping_rend = &_adjust_right_overlap_boundary_via_fuzzlength($overlapping_rend, $self, $other_alignment, $fuzzlength); + print "\tadjusted overlapping coords to: $overlapping_lend to $overlapping_rend\n" if $SEE; + } + + ## find introns within adjusted overlap range and ensure identity + my @a_intron_coords = $self->get_intron_coords(); + my @b_intron_coords = $other_alignment->get_intron_coords(); + + my @a_introns_in_range = &_get_overlapping_capped_introns($overlapping_lend, $overlapping_rend, \@a_intron_coords); + my @b_introns_in_range = &_get_overlapping_capped_introns($overlapping_lend, $overlapping_rend, \@b_intron_coords); + + if (@a_introns_in_range || @b_introns_in_range) { + ## ensure identity: + my %all_introns; + my %a_introns; + foreach my $coordset (@a_introns_in_range) { + my $key = join (",", @$coordset); + $a_introns{$key} = 1; + $all_introns{$key} = 1; + } + my %b_introns; + foreach my $coordset (@b_introns_in_range) { + my $key = join (",", @$coordset); + $b_introns{$key} = 1; + $all_introns{$key} = 1; + } + foreach my $intron_key (keys %all_introns) { + unless ($a_introns{$intron_key} && $b_introns{$intron_key}) { + return (0); # not compatible, an intron difference in the overlapping region exists. + } + } + } + + } + + ## if got this far, passed all compatibility tests. + + return (1); # yes, compatible. +} + +#### +sub _get_contained_coords { + my ($lend, $rend, $coordsets_aref) = @_; + my @contained_coords; + foreach my $coordset (@$coordsets_aref) { + my ($coord_lend, $coord_rend) = sort {$a<=>$b} @$coordset; + if ($lend <= $coord_lend && $coord_rend <= $rend) { + ## coordset contained + push (@contained_coords, $coordset); + } + } + return (@contained_coords); +} + +#### +# get introns that overlap lend and rend, and set termini of intron coords to these values if they extend beyond them. +sub _get_overlapping_capped_introns { + my ($lend, $rend, $coordsets_aref) = @_; + my @contained_coords; + foreach my $coordset (@$coordsets_aref) { + my ($coord_lend, $coord_rend) = sort {$a<=>$b} @$coordset; + if ($lend <= $coord_rend && $rend >= $coord_lend) { + ## coordset contained + if ($coord_lend < $lend) { + $coord_lend = $lend; + } + if ($coord_rend > $rend) { + $coord_rend = $rend; + } + push (@contained_coords, [$coord_lend, $coord_rend]); + } + } + return (@contained_coords); +} + + +#### +sub _adjust_left_overlap_boundary_via_fuzzlength { + my ($overlapping_lend, $alignment_a, $alignment_b, $fuzzlength) = @_; + + ## make adjustments to take into account fuzzlength and existing intron coordinates and adjacent segments + ## we trust short aligment segments that precede an intron, even if they're shorter than the fuzzlength + + my $a_overlapping_segment = $alignment_a->find_segment_containing_coord($overlapping_lend); + my $b_overlapping_segment = $alignment_b->find_segment_containing_coord($overlapping_lend); + my @bounds = (); + if ($a_overlapping_segment) { + my ($a_seg_lend, $a_seg_rend) = $a_overlapping_segment->get_coords(); + push (@bounds, $a_seg_rend); + } + if ($b_overlapping_segment) { + my ($b_seg_lend, $b_seg_rend) = $b_overlapping_segment->get_coords(); + push (@bounds, $b_seg_rend); + } + if (@bounds) { + my $max_bound = max_coord(@bounds); + my $delta = $max_bound - $overlapping_lend; + if ($delta < 0) { + confess "Error, delta left bound is less than zero"; + } + my $fuzz_employed = min_coord($fuzzlength, $delta); + $overlapping_lend += $fuzz_employed; + } + return ($overlapping_lend); +} + +#### +sub _adjust_right_overlap_boundary_via_fuzzlength { + my ($overlapping_rend, $alignment_a, $alignment_b, $fuzzlength) = @_; + + ## make adjustments to take into account fuzzlength and existing intron coordinates and adjacent segments + ## we trust short aligment segments that precede an intron, even if they're shorter than the fuzzlength + + my $a_overlapping_segment = $alignment_a->find_segment_containing_coord($overlapping_rend); + my $b_overlapping_segment = $alignment_b->find_segment_containing_coord($overlapping_rend); + my @bounds = (); + if ($a_overlapping_segment) { + my ($a_seg_lend, $a_seg_rend) = $a_overlapping_segment->get_coords(); + push (@bounds, $a_seg_lend); + } + if ($b_overlapping_segment) { + my ($b_seg_lend, $b_seg_rend) = $b_overlapping_segment->get_coords(); + push (@bounds, $b_seg_lend); + } + if (@bounds) { + my $min_bound = min_coord(@bounds); + my $delta = $overlapping_rend - $min_bound; + if ($delta < 0) { + confess "Error, delta left bound is less than zero"; + } + my $fuzz_employed = min_coord($fuzzlength, $delta); + $overlapping_rend -= $fuzz_employed; + } + return ($overlapping_rend); +} + + + + + +#### +sub find_segment_containing_coord { + my $self = shift; + my ($coord) = @_; + my @segments = $self->get_alignment_segments(); + foreach my $segment (@segments) { + my ($lend, $rend) = $segment->get_coords(); + if ($lend <= $coord && $coord <= $rend) { + return ($segment); + } + } + return (undef); # none found +} + + +sub _overlap { + my ($a1_lend, $a1_rend, $a2_lend, $a2_rend) = @_; + #print "Checking overlap @_\t"; + if ($a2_rend >= $a1_lend && $a2_lend <= $a1_rend) { #overlap + #print "YES\n"; + return (1); + } else { + #print "NO\n"; + return (0); + } +} + +=over 4 + +=item encapsulates() + +B Returns true (1) if this alignment encapsulates the span of the other alignment object + +B ($other_alignment_obj, $fuzz_dist) + +Alignments A and B are of type CDNA::CDNA_alignment + +B [1|0] + +=back + +=cut + + +sub encapsulates { + my $self = shift; + my ($alignmentB, $fuzz_dist) = @_; + my $alignmentA = $self; + + if (! defined $fuzz_dist) { + $fuzz_dist = 0; + } + + my ($alend, $arend) = $alignmentA->get_coords(); + my ($blend, $brend) = $alignmentB->get_coords(); + + if ($blend + $fuzz_dist >= $alend && + $brend - $fuzz_dist <= $arend) + { + return (1); + } else { + return (0); + } +} + +#### +sub _get_coords_of_overlap { + my ($a_lend, $a_rend, $b_lend, $b_rend) = @_; + unless (&_overlap($a_lend, $a_rend, $b_lend, $b_rend)) { + confess "Error, trying to get coordinates of overlapping region for two features that do not overlap!"; + } + my $overlapping_lend = &max_coord($a_lend, $b_lend); + my $overlapping_rend = &min_coord($a_rend, $b_rend); + + return ($overlapping_lend, $overlapping_rend); +} + +#### +sub min_coord { + my @coords = @_; + @coords = sort {$a<=>$b} @coords; + my $min_coord = shift @coords; + return ($min_coord); +} + +#### +sub max_coord { + my @coords = @_; + @coords = sort {$a<=>$b} @coords; + my $max_coord = pop @coords; + return ($max_coord); +} + + +1; #EOM + + + + + diff --git a/99.scripts/trinity_utils/PerlLib/CDNA/CDNA_stitcher.pm b/99.scripts/trinity_utils/PerlLib/CDNA/CDNA_stitcher.pm new file mode 100644 index 0000000..01dafaa --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/CDNA/CDNA_stitcher.pm @@ -0,0 +1,289 @@ +package main; +our $SEE; + +package CDNA::CDNA_stitcher; +use strict; +use CDNA::CDNA_alignment; +use CDNA::Gene_obj_alignment_assembler; #used for converting gene_obj to alignment. +use Carp; + +my $FUZZDIST = 20; #allow FUZZDIST nt extension beyond splice site. + +sub new { + my $packagename = shift; + my $self = { + fuzzlength=>$FUZZDIST #default setting. + }; + bless ($self, $packagename); + return ($self); +} + + +sub set_fuzzlength { + my $self = shift; + my $fuzzdist = shift; + $self->{fuzzlength} = $fuzzdist; +} + +sub stitch_alignments { + my $self = shift; + my ($template_alignment, $thread_alignment) = @_; + + my $template_acc = $template_alignment->get_acc(); + my $thread_acc = $thread_alignment->get_acc(); + + print "Stitching Template:\n$template_acc: " . $template_alignment->toToken() + . "\nby\n$thread_acc: " . $thread_alignment->toToken() . "\n\n" if $main::SEE; + + ## don't even think about stitching alignments of opposite spliced orienatations!!! + + my $template_spliced_orient = $template_alignment->get_spliced_orientation(); + my $thread_spliced_orient = $thread_alignment->get_spliced_orientation(); + + if ($template_spliced_orient ne '?' && $thread_spliced_orient ne '?') { + ## both have assigned transcribed orients + if ($template_spliced_orient ne $thread_spliced_orient) { + confess "Error, trying to stitch together oppositely transcribed alignments. That's impossible. "; + } + } + + + my $spliced_orient = ($thread_spliced_orient ne '?') ? $thread_spliced_orient : $template_spliced_orient; + + my $fuzzlength = $self->{fuzzlength}; + + ## Algorithm + ## -anchor the first thread segment to the template alignment + ## -determine the boundaries of the matching anchor and first threaded segment + ## -add all preceding template segments to the stitched alignment if splice compatible. + ## -add internal cdna alignment segments + ## -add terminal template alignment segments if splice compatible. + + my @template_segments = $template_alignment->get_alignment_segments(); + my @threaded_segments = $thread_alignment->get_alignment_segments(); + + my $single_threaded_segment = ($#threaded_segments == 0) ? 1:0; #indicate only a single segment exists. + + my @stitched_segments; + ## Anchor first threaded segment to the template alignment: + my ($thread_lend, $thread_rend) = $threaded_segments[0]->get_coords(); + my $anchor_pos = -1; + for (my $i = 0; $i <= $#template_segments; $i++) { + my ($lend, $rend) = $template_segments[$i]->get_coords(); + if ($lend < $thread_rend && $rend > $thread_lend) { #overlap + $anchor_pos = $i; + last; + } + } + print "Anchoring first cDNA segment. Anchor point: $anchor_pos\n" if $SEE; + if ($anchor_pos > -1) { #found an anchor point in template alignment. + + + ## add stitched segment + my $anchor_segment = $template_segments[$anchor_pos]; + my ($anchor_lend, $anchor_rend) = $anchor_segment->get_coords(); + my $first_threaded_segment = $threaded_segments[0]; + my ($thread_lend, $thread_rend) = $first_threaded_segment->get_coords(); + # initialize to largest spread + my ($new_lend) = ($anchor_lend < $thread_lend) ? $anchor_lend : $thread_lend; + my ($new_rend) = ($anchor_rend > $thread_rend) ? $anchor_rend : $thread_rend; + # adjust based on splice boundaries + my ($has_left_splice_junction, $has_right_splice_junction); + $has_left_splice_junction = $anchor_segment->has_left_splice_junction(); #initialize. + if ($anchor_segment->has_left_splice_junction() && ($anchor_lend - $thread_lend <= $fuzzlength)) { + $new_lend = $anchor_lend; + } elsif ($thread_lend < $anchor_lend) { + #using threaded segment as initial stitched segment + $has_left_splice_junction = $first_threaded_segment->has_left_splice_junction(); + } + + $has_right_splice_junction = $first_threaded_segment->has_right_splice_junction(); + if ($first_threaded_segment->has_right_splice_junction()) { + $new_rend = $thread_rend; + } elsif ($anchor_segment->has_right_splice_junction() && ($thread_rend - $anchor_rend <= $fuzzlength)) { + $new_rend = $anchor_rend; + $has_right_splice_junction = $anchor_segment->has_right_splice_junction(); + } + + ## Add preceding template segments if current segment is left-splice compatible. + if ($has_left_splice_junction) { + print "Adding all template segments before position $anchor_pos.\n" if $SEE; + ## add all template segments to the stitched alignment preceding the anchor point: + for (my $i = 0; $i < $anchor_pos; $i++) { + push (@stitched_segments, $template_segments[$i]->clone()); + } + } + + ## add the current segment. + my $newsegment = $anchor_segment->clone(); + $newsegment->set_coords($new_lend, $new_rend); + $newsegment->set_left_splice_junction($has_left_splice_junction); + $newsegment->set_right_splice_junction($has_right_splice_junction); + print "adding current segment: lsplice: $has_left_splice_junction, rsplice: $has_right_splice_junction\n" if $SEE; + push (@stitched_segments, $newsegment); + + } else { #no anchor pos, simply add the first threaded segment: + print "No anchor position, simply adding the first threaded segment.\n" if $SEE; + push (@stitched_segments, $threaded_segments[0]->clone()); + } + + ## Add all but last threaded segments to the stitched alignment: + + for (my $i = 1; $i < $#threaded_segments; $i++) { + print "Adding internal cDNA alignment segment, index: $i\n" if $SEE; + push (@stitched_segments, $threaded_segments[$i]->clone()); + } + + ## Anchor the last threaded segment to a segment within the template model: + print "Anchoring the terminal segment.\n" if $SEE; + $anchor_pos = -1; + my $last_threaded_segment = $threaded_segments[$#threaded_segments]; + ($thread_lend, $thread_rend) = $last_threaded_segment->get_coords(); + # find last template segment overlapping last cdna segment + for (my $i = $#template_segments; $i >= 0; $i--) { + my ($lend, $rend) = $template_segments[$i]->get_coords(); + if ($lend < $thread_rend && $rend > $thread_lend) { #overlap + $anchor_pos = $i; + last; + } + } + print "Terminal anchor position: $anchor_pos\n" if $SEE; + if ($anchor_pos > -1) { + #build composite segment + my $anchor_segment = $template_segments[$anchor_pos]; + my $has_left_splice_junction = $last_threaded_segment->has_left_splice_junction(); #initialize. + my $has_right_splice_junction = $anchor_segment->has_right_splice_junction(); + my ($anchor_lend, $anchor_rend) = $anchor_segment->get_coords(); + my $new_lend = ($anchor_lend < $thread_lend) ? $anchor_lend : $thread_lend; + my $new_rend = ($anchor_rend > $thread_rend) ? $anchor_rend : $thread_rend; + if ($last_threaded_segment->has_left_splice_junction()) { + $new_lend = $thread_lend; + } elsif ($anchor_segment->has_left_splice_junction() && ($anchor_lend - $thread_lend <= $fuzzlength)) { + $new_lend = $anchor_lend; + $has_left_splice_junction = $anchor_segment->has_left_splice_junction(); + } + if ($anchor_segment->has_right_splice_junction() && ($thread_rend - $anchor_rend <= $fuzzlength)) { + $new_rend = $anchor_rend; + } else { + $has_right_splice_junction = $last_threaded_segment->has_right_splice_junction(); + } + if ($single_threaded_segment) { #only one segment (first == last) + #just update existing status: + print "Only one threaded segment, updating it's rend coord and status.\n" if $SEE; + my $last_stitched_segment = $stitched_segments[$#stitched_segments]; + my ($curr_lend, $curr_rend) = $last_stitched_segment->get_coords(); + $last_stitched_segment->set_coords($curr_lend, $new_rend); + $last_stitched_segment->set_right_splice_junction($has_right_splice_junction); + } else { + my $newsegment = $anchor_segment->clone(); + $newsegment->set_coords($new_lend, $new_rend); + $newsegment->set_left_splice_junction($has_left_splice_junction); + $newsegment->set_right_splice_junction($has_right_splice_junction); + print "adding composite terminal segment. Lsplice: $has_left_splice_junction, Rsplice: $has_right_splice_junction.\n" if $SEE; + push (@stitched_segments, $newsegment); + } + } else { + print "No matching terminal exon in template.\n" if $SEE; + if ($single_threaded_segment) { + print "Single threaded segment remains unchanged.\n" if $SEE; + } else { + + print "Adding last cDNA exon.\n" if $SEE; + push (@stitched_segments, $last_threaded_segment->clone()); + } + } + + + ## Add terminal template segments if the last stitched segment contains a splice junction + + my $last_stitched_segment = $stitched_segments[$#stitched_segments]; + if ($last_stitched_segment->has_right_splice_junction()) { + print "Adding terminal template segments:\n" if $SEE; + my ($last_lend, $last_rend) = $last_stitched_segment->get_coords(); + $anchor_pos = -1; + for (my $i = $#template_segments; $i >= 0; $i--) { + my ($anchor_lend, $anchor_rend) = $template_segments[$i]->get_coords(); + if ($anchor_rend > $last_lend && $anchor_lend < $last_rend) { #overlap + $anchor_pos = $i; + last; + } + } + print "Last overlapping alignment segment in index found at: $anchor_pos\n" if $SEE; + if ($anchor_pos > -1) { + for (my $i = $anchor_pos + 1; $i <= $#template_segments; $i++) { + print "\tadding terminal template segment index: $i\n" if $SEE; + push (@stitched_segments, $template_segments[$i]->clone()); + } + } + } + + + ## Done stitching segments together. Create new alignment: + my $cdna_length = 0; + foreach my $segment (@stitched_segments) { + my $length = $segment->get_length(); + $cdna_length += $length; + } + + my $seq_ref = $template_alignment->get_genomic_seq_ref(); + + my $stitched_alignment = new CDNA::CDNA_alignment ($cdna_length, \@stitched_segments, $seq_ref); #aligned orientation auto assigned. + + $stitched_alignment->set_spliced_orientation($spliced_orient); + $stitched_alignment->set_acc($template_alignment->get_acc()); + + print "Stitched alignment: " . $stitched_alignment->toToken() . "\n" if $SEE; + + return ($stitched_alignment); +} + + + +sub stitch_alignment_into_gene { + my $self = shift; + my ($gene_obj, $cdna_obj, $sequence_ref) = @_; + + my $gene_obj_assembler = new CDNA::Gene_obj_alignment_assembler($sequence_ref); + my $gene_based_cdna_obj = $gene_obj_assembler->gene_obj_to_cdna_alignment($gene_obj); + + my $stitched_alignment = $self->stitch_alignments($gene_based_cdna_obj, $cdna_obj); + $stitched_alignment->set_fli(1); + + my $partials_href = $self->_analyze_partial_status($gene_obj); + my $stitched_gene_obj = $stitched_alignment->get_gene_obj_via_alignment($partials_href); + + return ($stitched_gene_obj); + +} + + +#### +sub _analyze_partial_status { + my $self = shift; + my ($gene_obj) = shift; + my %partials; + if ($gene_obj->is_5prime_partial()) { + $partials{"5prime"} = 1; + } + if ($gene_obj->is_3prime_partial()) { + $partials{"3prime"} = 1; + } + return (\%partials); +} + + + + +1; #EOM + + + + + + + + + + + + diff --git a/99.scripts/trinity_utils/PerlLib/CDNA/Gene_obj_alignment_assembler.pm b/99.scripts/trinity_utils/PerlLib/CDNA/Gene_obj_alignment_assembler.pm new file mode 100644 index 0000000..99b06c9 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/CDNA/Gene_obj_alignment_assembler.pm @@ -0,0 +1,140 @@ +package main; +our $SEE; + +=head1 NAME + +CDNA::Gene_obj_alignment_assembler; + +=cut + +=head1 DESCRIPTION + +This package assembles multiple gene objs by assembling compatible sets of coordinates. This module inherits from the CDNA::Genome_based_cDNA_assembler module. Please see this inherited module for additional methods available. + +=cut + +package CDNA::Gene_obj_alignment_assembler; +use strict; +use base qw (CDNA::PASA_alignment_assembler); +use CDNA::CDNA_alignment; + + +=item new() + +=over 4 + +B Instantiate a new CDNA::Gene_obj_alignment_assembler object. + +B $seq_ref + +$seq_ref is a scalar reference to the genomic sequence string. + +B obj_ref + +=back + +=cut + + +sub new { + my $packagename = shift; + my $seqref = shift; + my $self = $packagename->SUPER::new(); + $self->{sequence_ref} = $seqref; + bless ($self, $packagename); + + return ($self); +} + + + +=item assemble_genes() + +=over 4 + +B assembles a list of Gene_obj(s). Each of the Gene_objs + +B @Gene_obj + +@Gene_obj is an array of gene objects created via the Gene_obj.pm module. + +B @CDNA_alignments + +The @CDNA_alignments array contains the list of CDNA::CDNA_alignment objects created based on the gene objects. + +The CDNA_alignment accession is set to the Model_feat_name of the gene_obj. Retrieving the cDNA acc using the get_acc() method will allow a mapping back to the gene object. Also, the newly created CDNA_alignment objects that are returned should be in the same order as the inputed gene objects, so indexing should provide mapping info as well. + +use the get_assemblies() function to fetch the merged genes. + +=back + +=cut + + + + + +sub assemble_genes { + my $self = shift; + my @gene_objs = @_; + my @cDNA_alignments; + foreach my $gene_obj (@gene_objs) { + my $alignment_obj = $self->gene_obj_to_cdna_alignment($gene_obj); + push (@cDNA_alignments, $alignment_obj); + } + $self->assemble_alignments(@cDNA_alignments); + return (@cDNA_alignments); +} + + + +=item gene_obj_to_cdna_alignment() + +=over 4 + +B method converts a Gene_obj to a CDNA::CDNA_alignment obj. + +B Gene_obj + +B CDNA::CDNA_alignment + +=back + +=cut + +sub gene_obj_to_cdna_alignment { + my $self = shift; + my $gene_obj = shift; + my @exons = $gene_obj->get_exons(); + my %coords; + my @alignment_segments; + my $cdna_length = 0; + foreach my $exon (@exons) { + my ($end5, $end3) = $exon->get_coords(); + my $alignment_seg = new CDNA::Alignment_segment($end5, $end3); + push (@alignment_segments, $alignment_seg); + $cdna_length += abs ($end3 - $end5) + 1; + } + my $alignment_obj = new CDNA::CDNA_alignment($cdna_length, \@alignment_segments, $self->{sequence_ref}); + my $feat_name = $gene_obj->{Model_feat_name}; + print "Gene has the following model feat_name: $feat_name\n" if $SEE; + $alignment_obj->set_acc($feat_name); + $alignment_obj->set_title($gene_obj->{com_name}); + $alignment_obj->set_fli_status(1); #treat like a fli cdna + my $align_orient = $alignment_obj->get_spliced_orientation(); + ## Make sure strand is set correctly (should only matter for single exon genes). + if ($align_orient ne $gene_obj->{strand}) { + $alignment_obj->set_orientation($gene_obj->{strand}); + $alignment_obj->set_spliced_orientation($gene_obj->{strand}); + } + $alignment_obj->remap_cdna_segment_coords(); + print $alignment_obj->toToken() ." " . $alignment_obj->get_acc() . "\n" if $SEE; + return ($alignment_obj); +} + + +1; #EOM + + + + diff --git a/99.scripts/trinity_utils/PerlLib/CDNA/Genome_based_cDNA_assembler.pm b/99.scripts/trinity_utils/PerlLib/CDNA/Genome_based_cDNA_assembler.pm new file mode 100644 index 0000000..27beb53 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/CDNA/Genome_based_cDNA_assembler.pm @@ -0,0 +1,630 @@ +#!/usr/local/bin/perl + +package main; +our $SEE; + + +package CDNA::Genome_based_cDNA_assembler; + +=head1 NAME + +CDNA::Genome_based_cdna_assembler + +=cut + +=head1 DESCRIPTION + +This module is used to assemble compatible cDNA alignments. The algorithm is as follows: +must describe this here. + +=cut + +use strict; +use CDNA::CDNA_alignment; +use Data::Dumper; + +my $DELIMETER = "$;,"; +my $FUZZLENGTH = 20; + + +=item new() + +=over 4 + +B instantiates a new cDNA assembler obj. + +B $sequence_sref + +$sequence_sref is a reference to a scalar containing the genomic sequence string. + +B $obj_href + +$obj_href is the object reference newly instantiated by this new method. + +=back + +=cut + +sub new () { + my $package_name = shift; + my $self = {}; + bless ($self, $package_name); + $self->_init(@_); + return ($self); +} + + +sub _init { + my $self = shift; + my ($sequence_ref) = @_; + $self->{incoming_alignments} = []; #these are the alignments to be assembled. + $self->{assemblies} = []; #contains list of all singletons and assemblies. + $self->{sequence_ref} = $sequence_ref; + $self->{fuzzlength} = $FUZZLENGTH; #default setting. +} + + +=item assemble_alignments() + +=over 4 + +B assembles a series of cDNA aligmnments into one or more cDNA assemblies + +B @alignments + +@alignments is an array of CDNA::CDNA_alignment objects + +B none. + +=back + +=cut + +sub assemble_alignments { + my $self = shift; + my @alignments = @_; + @alignments = reverse sort {$a->{length}<=>$b->{length}} @alignments; #keep in order of decreasing alignment length. + $self->{incoming_alignments} = [@alignments]; + #return; + + ## Algorithm: given cdna, merge with rest of cDNAs iteratively until merging complete. + ## find unmerged cDNA, merge it if possible with all individual cdna entries. + ## continue until all cDNAs have merged products, if possible. + + my $num_alignments = $#alignments + 1; + for (my $i = 0; $i < $num_alignments; $i++) { + my $seed_alignment = $alignments[$i]; + if ($seed_alignment->{merged}) {next;} + print "Checking seed $i\n" if $SEE; + my $merged_alignment = $seed_alignment; #initialize to seed alignment. + my $merged_flag = 1; + my $round = 0; + while ($merged_flag) { + $merged_flag = 0; + $round++; + print "Merging round $round\n" if $SEE; + for (my $j=0; $j < $num_alignments; $j++) { + if ($i == $j) {next;} #no self comparisons. + my $other_alignment = $alignments[$j]; + if ($self->already_contains($merged_alignment, $other_alignment)) { next;} + if ($self->can_merge($merged_alignment, $other_alignment)) { + my $initial_merged_fli_status = $merged_alignment->is_fli(); + my $initial_merged_orient = $merged_alignment->get_orientation(); + $merged_alignment = $self->merge_alignments($merged_alignment, $other_alignment); + $alignments[$i]->{merged} = 1; #set seed alignment merge flag. + $alignments[$j]->{merged} = 1; #set other alignment merge flag. + $merged_flag = 1; #indicates something actually merged this round. + + $merged_alignment->remap_cdna_segment_coords(); + print "merged: " . $merged_alignment->toToken() . "\n" if $SEE; + + } + } + } + push (@{$self->{assemblies}}, $merged_alignment); #either an assembled product, or something that won't ever merge. + } +} + +=item get_assemblies() + +=over 4 + +B returns all the alignment assemblies resulting from the assembly procedure. + +B none. + +B @assemblies + +@assemblies is an array of CDNA::CDNA_alignment objects. + +use the get_acc() method of the alignment object to retrieve all the accessions of the cDNAs that were merged into the assembly. + +=back + +=cut + + +sub get_assemblies { + my $self = shift; + return (@{$self->{assemblies}}); +} + + +# private method. Determines if two alignments are compatible with one another. +sub can_merge () { + my $self = shift; + my ($a1, $a2) = @_; + print "Checking to see if can merge: " . $a1->get_acc() . ", " . $a2->get_acc() . "\n" if $::SEE; + ## See if the coord spans overlap + my ($a1_lend, $a1_rend) = $a1->get_coords(); + my ($a2_lend, $a2_rend) = $a2->get_coords(); + unless (&overlap($a1_lend, $a1_rend, $a2_lend, $a2_rend)) { + print "failed merge: No overlap between alignment spans. ($a1_lend, $a1_rend) vs. ($a2_lend, $a2_rend)\n" if $::SEE; + return(0); + } + + ## Make sure the spliced orientation is equivalent if appropriate + my $a1_num_segs = $a1->get_num_segments(); + my $a2_num_segs = $a2->get_num_segments(); + my $a1_spliced_orientation = $a1->get_spliced_orientation(); + my $a2_spliced_orientation = $a2->get_spliced_orientation(); + my $a1_is_fli = $a1->is_fli(); + my $a2_is_fli = $a2->is_fli(); + + my $fuzzlength = $self->{fuzzlength}; + + if ($a1_num_segs > 1 && $a2_num_segs > 1) { #if more than one segment, then spliced orientation is relevant. + if ($a1_spliced_orientation ne $a2_spliced_orientation) { + print "failed merge: $a1_num_segs segments vs. $a2_num_segs and opposite spliced orientations.\n" if $::SEE; + return (0); + } + } + + if ($a1_is_fli && $a2_is_fli && ($a1_spliced_orientation ne $a2_spliced_orientation)) { #fli's must have same orient. + print "failed merge: (a1-fli: $a1_is_fli, a2-fli: $a2_is_fli) and opposite orientations.\n" if $SEE; + return (0); + } + + ## Check all overlapping segments to ensure non-conflicting segments. + my @a1_segments = $a1->get_alignment_segments(); + my @a2_segments = $a2->get_alignment_segments(); + + + ## align segment orders between a1 and a2 + my ($starting_a1, $starting_a2); + for (my $i = 0; $i <= $#a1_segments; $i++) { + my $a1_seg = $a1_segments[$i]; + my ($a1_lend, $a1_rend) = $a1_seg->get_coords(); + for (my $j = 0; $j <= $#a2_segments; $j++) { + my $a2_seg = $a2_segments[$j]; + my ($a2_lend, $a2_rend) = $a2_seg->get_coords(); + if (&overlap($a1_lend, $a1_rend, $a2_lend, $a2_rend)) { + $starting_a1 = $i; + $starting_a2 = $j; + last; + } + } + if (defined ($starting_a1) && defined ($starting_a2)) { + last; + } + } + + unless (defined ($starting_a1) && defined ($starting_a2)) { + print "failed merge: can't align two segments between overlapping alignments.\n" if $::SEE; + return (0); + } + + unless ($starting_a1 == 0 || $starting_a2 == 0) { + print "failed merge: segment alignment doesn't begin at either cDNA terminus.\n" if $::SEE; + return (0); + } + + while ($starting_a1 <= $#a1_segments && $starting_a2 <= $#a2_segments) { + my $a1_segment = $a1_segments[$starting_a1]; + my $a2_segment = $a2_segments[$starting_a2]; + + my ($a1_lend, $a1_rend) = $a1_segment->get_coords(); + my ($a2_lend, $a2_rend) = $a2_segment->get_coords(); + + if (&overlap($a1_lend, $a1_rend, $a2_lend, $a2_rend)) { + ## See if have splice sites, do they exist and are they identical + if ($a1_segment->has_left_splice_junction() || $a2_segment->has_left_splice_junction()) { + if ($a1_segment->has_left_splice_junction() && $a2_segment->has_left_splice_junction() && $a1_lend != $a2_lend) { + print "failed merge:\tboth left splice, but unequal coords: L1 ($a1_lend), L2 ($a2_lend)\n" if $::SEE; + return (0); + } elsif ($a1_segment->has_left_splice_junction() && ($a2_lend + $fuzzlength < $a1_lend)) { #alignment extends beyond a splice junction. + print "failed merge:\tL1 left splice, L2 ($a2_lend) < L1 ($a1_lend)\n" if $::SEE; + return (0); + } elsif ($a2_segment->has_left_splice_junction() && ($a1_lend + $fuzzlength < $a2_lend)) { + print "failed merge:\tL2 left splice, L1 ($a1_lend) < L2 ($a2_lend)\n" if $::SEE; + return (0); + } + + } + if ($a1_segment->has_right_splice_junction() || $a2_segment->has_right_splice_junction()) { + if ($a1_segment->has_right_splice_junction() && $a2_segment->has_right_splice_junction() && $a1_rend != $a2_rend) { + print "failed merge:\tboth right splice, but unequal coords: R1($a1_rend), R2($a2_rend)\n" if $::SEE; + return (0); + } elsif ($a1_segment->has_right_splice_junction() && ($a2_rend - $fuzzlength > $a1_rend)) { + print "failed merge:\tR1 right splice, R2($a2_rend) > R1 ($a1_rend)\n" if $::SEE; + return (0); + } elsif ($a2_segment->has_right_splice_junction() && ($a1_rend - $fuzzlength > $a2_rend)) { + print "failed merge:\tR2 right splice, R1 ($a1_rend) > R2 ($a2_rend)\n" if $::SEE; + return (0); + } + } + } else { + print "failed merge: Two ordered segments don't overlap. ($a1_lend, $a1_rend) , ($a2_lend, $a2_rend)\n" if $::SEE; + return (0); + } + $starting_a1++; + $starting_a2++; + + } + + ## Passed all tests + print "Merge tests PASSED.\n" if $::SEE; + return (1); + +} + + + +# private method +# merges two alignment objects together into an assembly. +sub merge_alignments () { + my ($self, $a1, $a2) = @_; + print "Merging <" . $a1->get_acc() . ">, <" . $a2->get_acc() . ">\n" if $::SEE; + my $a1_fli = $a1->is_fli(); + my $a2_fli = $a2->is_fli(); + print "a1_fli: $a1_fli, a2_fli: $a2_fli\n" if $SEE; + ## Determine fli status for merged product. + my $merged_fli_status = ($a1_fli || $a2_fli); + + my $merged_orientation = $self->determine_merged_orientation($a1, $a2); + + #get a1 segment cooridnates; + my %leftsplicecoords; #preferrentially use splice coords over Fuzzlength extensions. + my %rightsplicecoords; + my %a1_coords; + my @a1_segments = $a1->get_alignment_segments(); + foreach my $seg (@a1_segments) { + my ($lend, $rend) = $seg->get_coords(); + $a1_coords{$lend} = $rend; + if ($seg->has_left_splice_junction()) { + $leftsplicecoords{$lend} = 1; + } + if ($seg->has_right_splice_junction()) { + $rightsplicecoords{$rend} = 1; + } + } + + # get a2 segment coordinates: + my %a2_coords; + my @a2_segments = $a2->get_alignment_segments(); + foreach my $seg (@a2_segments) { + my ($lend, $rend) = $seg->get_coords(); + $a2_coords{$lend} = $rend; + if ($seg->has_left_splice_junction()) { + $leftsplicecoords{$lend} = 1; + } + if ($seg->has_right_splice_junction()) { + $rightsplicecoords{$rend} = 1; + } + } + + my %merged_coords; + + #print "Coord dumps:\n" . Dumper (\%a1_coords) . Dumper (\%a2_coords) . "\n"; + + #print "\n\nBEGIN\n"; + ## merge the overlapping coordinate sets between a1 and a2 alignments. + foreach my $a1_lend (keys %a1_coords) { + my $a1_rend = $a1_coords{$a1_lend}; + my ($merged_lend, $merged_rend); + foreach my $a2_lend (keys %a2_coords) { + my $a2_rend = $a2_coords{$a2_lend}; + if (&overlap($a1_lend, $a1_rend, $a2_lend, $a2_rend)) { #overlap + ## Determine merged lend; + if ($leftsplicecoords{$a1_lend}) { + $merged_lend = $a1_lend; + } elsif ($leftsplicecoords{$a2_lend}) { + $merged_lend = $a2_lend; + } else { + $merged_lend = min ($a1_lend, $a2_lend); + } + # Determine merged rend + if ($rightsplicecoords{$a1_rend}) { + $merged_rend = $a1_rend; + } elsif ($rightsplicecoords{$a2_rend}) { + $merged_rend = $a2_rend; + } else { + $merged_rend = max ($a1_rend, $a2_rend); + } + last; + } + } + if ($merged_lend && $merged_rend) { + #print "Adding overlapped \$merged_coords{$merged_lend} = $merged_rend\n"; + $merged_coords{$merged_lend} = $merged_rend; + } else { #must not have been any overlap; keep a1 coordset + #print "Keeping a1 coords: \$merged_coords{$a1_lend} = $a1_rend\n"; + $merged_coords{$a1_lend} = $a1_rend; + } + } + #print "adding unconsumed a2 coords:\n"; + ## add non-overlapping a2-segments + foreach my $a2_lend (keys %a2_coords) { + my $overlap = 0; + my $a2_rend = $a2_coords{$a2_lend}; + foreach my $m_lend (keys %merged_coords) { + my $m_rend = $merged_coords{$m_lend}; + if (&overlap($a2_lend, $a2_rend, $m_lend, $m_rend)) { + $overlap = 1; + last; + } + } + if (!$overlap) { + #print "Consuming a2 non-overlapping coords.\n"; + $merged_coords{$a2_lend} = $a2_rend; #added a2 non-overlapping coordset + } else { + #print "coords overlapped, not consuming.\n"; + } + } + print Dumper (\%merged_coords) if $::SEE; + ## Create a new alignment based on a1 and a2 + my @alignment_segments; + my $merged_length = 0; + foreach my $end5 (keys %merged_coords) { + my $end3 = $merged_coords{$end5}; + my $alignment_seg = new CDNA::Alignment_segment($end5, $end3); + push (@alignment_segments, $alignment_seg); + $merged_length += abs ($end3 - $end5) + 1; + } + my $new_alignment = new CDNA::CDNA_alignment($merged_length, \@alignment_segments, $self->{sequence_ref}); + my $new_acc = $self->merge_accs($a1->get_acc(), $a2->get_acc()); + $new_alignment->set_acc($new_acc); + $new_alignment->set_fli_status($merged_fli_status); + $new_alignment->force_spliced_validation($merged_orientation); + #print "END.\n\n"; + return ($new_alignment); +} + + +#private method +# returns the minimum of an array of numerical values. +sub min { + my @x = @_; + @x = sort {$a<=>$b} @x; + my $y = shift @x; + return ($y); +} + + +#private method +# returns the maximum of an array of numerical values. +sub max { + my @x = @_; + @x = sort {$a<=>$b} @x; + my $y = pop @x; + return ($y); +} + + + +#private method +# returns true/false, determines whether two coordinate sets overlap each other. +sub overlap { + my ($a1_lend, $a1_rend, $a2_lend, $a2_rend) = @_; + #print "Checking overlap @_\t"; + if ($a2_rend >= $a1_lend && $a2_lend <= $a1_rend) { #overlap + #print "YES\n"; + return (1); + } else { + #print "NO\n"; + return (0); + } +} + + + + + +=item toAlignIllustration() + +=over 4 + +B illustrates the individual cDNAs to be assembled along with the final products. + +B $max_line_chars(optional) + +$max_line_chars is an integer representing the maximum number of characters in a single line of output to the terminal. The default is 100. + +B $alignment_illustration_text + +$alignment_illustration_text is a string containing a paragraph of text which illustrates the alignments and assemblies. An example is below: + + ---> <--> <-----> <---> <---------------- (+)gi|1199466 + + ---> <--> <-----> <---> <------------ (+)gi|1209702 + +----> <--> <---- (+)AV827070 + +----> <--> <--- (+)AV828861 + +----> <--> <--- (+)AV830936 + + ---> <--> <- (+)H36350 + +ASSEMBLIES: (1) + +----> <--> <-----> <---> <---------------- (+) gi|1199466, gi|1209702, AV827070, AV828861, AV830936, H36350 + + + + +=back + +=cut + + +sub toAlignIllustration () { + my $self = shift; + my $max_line_chars = shift; + $max_line_chars = ($max_line_chars) ? $max_line_chars : 100; #if not specified, 100 chars / line is default. + + ## Get minimum coord for relative positioning. + my @coords; + my @alignments = @{$self->{incoming_alignments}}; + foreach my $alignment (@alignments) { + my @c = $alignment->get_coords(); + push (@coords, @c); + } + @coords = sort {$a<=>$b} @coords; + print "coords: @coords\n" if $::SEE; + my $min_coord = shift @coords; + my $max_coord = pop @coords; + my $rel_max = $max_coord - $min_coord; + my $alignment_text = ""; + ## print each alignment followed by assemblies: + my $num_alignments = $#alignments + 1; + $alignment_text .= "Individual Alignments: ($num_alignments)\n"; + my $i = 0; + foreach my $alignment (@alignments) { + $alignment_text .= (sprintf ("%3d ", $i)) . $alignment->toAlignIllustration($min_coord, $rel_max, $max_line_chars) . "\n"; + $i++; + } + + my @assemblies = @{$self->{assemblies}}; + my $num_assemblies = $#assemblies + 1; + $alignment_text .= "\n\nASSEMBLIES: ($num_assemblies)\n"; + foreach my $assembly (@assemblies) { + $alignment_text .= " " . $assembly->toAlignIllustration($min_coord, $rel_max, $max_line_chars) . "\n"; + } + + return ($alignment_text); +} + + + +=over 4 + +=item set_fuzzlength() + +B Sets the fuzzlength parameter. + +B int + +B none. + +The fuzzlength is the length allowed to be fuzzy at the terminus of all alignments when compared to overlapping exons containining nearby splice sites. + +=back + +=cut + +sub set_fuzzlength { + my $self = shift; + my $fuzzlength = shift; + + $self->{fuzzlength} = $fuzzlength; +} + + + +#private method +# determines if two assemblies have the same accessions. +sub composition_same () { + my ($name1, $name2) = @_; + if ($name1 eq $name2) { + return (1); + } else { + return (0); + } +} + +#private method +# creates a new name based on the accessions composing two assemblies. +sub merge_accs () { + my $self = shift; + my @names = @_; + my @nameaccs; + foreach my $name (@names) { + my @nameacclist = split (/$DELIMETER/, $name); + push (@nameaccs, @nameacclist); + } + my %unique; + foreach my $acc (@nameaccs) { + $unique{$acc} = 1; + } + my @accs = sort {$a cmp $b} keys %unique; + my $combo = join ($DELIMETER, @accs); + return ($combo); +} + +#private method +# checks to see if an assembly already contains all the accessions built into a second assembly. +sub already_contains () { + ## Checks to see if a1 contains all accessions of a2 + my $self = shift; + my ($a1, $a2) = @_; + my $a1_acc = $a1->get_acc(); + my $a2_acc = $a2->get_acc(); + my @a1_accs = split (/$DELIMETER/, $a1_acc); + my @a2_accs = split (/$DELIMETER/, $a2_acc); + my %a1_acc_hash; + foreach my $acc (@a1_accs) { + $a1_acc_hash{$acc} = 1; + } + my %a2_acc_hash; + foreach my $acc (@a2_accs) { + $a2_acc_hash{$acc} = 1; + } + + foreach my $acc (keys %a2_acc_hash) { + unless ($a1_acc_hash{$acc}) { + return (0); + } + } + ## if still here, then must have all a2 accs in a1. + return (1); +} + + +#private +sub determine_merged_orientation { + my ($self, $a1, $a2) = @_; + ## Determine the orientation of the merged product. + my $num_a1_segments = $a1->get_num_segments(); + my $num_a2_segments = $a2->get_num_segments(); + my $a1_spliced_orientation = $a1->get_spliced_orientation(); + my $a2_spliced_orientation = $a2->get_spliced_orientation(); + print "a1_spliced_orient: $a1_spliced_orientation, a2_spliced_orient: $a2_spliced_orientation\n" if $SEE; + my $merged_orientation = '+'; #initialize to a default. + if ($a1_spliced_orientation eq $a2_spliced_orientation) { + $merged_orientation = $a1_spliced_orientation; + print "Same spliced orientation: $merged_orientation\n" if $SEE; + } elsif ($a1->is_fli() || $a2->is_fli()) { + $merged_orientation = ($a1->is_fli()) ? $a1_spliced_orientation : $a2_spliced_orientation; + print "a1 is fli, using a1 orient: $merged_orientation\n" if ($a1->is_fli() && $SEE); + print "a2 is fli, using a2 orient: $merged_orientation\n" if ($a2->is_fli() && $SEE); + + } elsif ($num_a1_segments > 1 || $num_a2_segments > 1) { + $merged_orientation = ($num_a1_segments > 1) ? $a1_spliced_orientation : $a2_spliced_orientation; + print "a1 has multiple segments, using a1 orient: $merged_orientation\n" if ($num_a1_segments > 1 && $SEE); + print "a2 has multiple segments, using a2 orient: $merged_orientation\n" if ($num_a2_segments > 1 && $SEE); + + } else { + print "Can't predict the orientation of the merged alignments ($a1_spliced_orientation vs. $a2_spliced_orientation). Using default '+'\n" if $SEE; + } + return ($merged_orientation); +} + + + +1; #EOM + + + + + diff --git a/99.scripts/trinity_utils/PerlLib/CDNA/Genome_based_cDNA_graph_assembler.pm b/99.scripts/trinity_utils/PerlLib/CDNA/Genome_based_cDNA_graph_assembler.pm new file mode 100644 index 0000000..5ee2e82 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/CDNA/Genome_based_cDNA_graph_assembler.pm @@ -0,0 +1,671 @@ +#!/usr/local/bin/perl + +package main; +our $SEE; + + +package CDNA::Genome_based_cDNA_graph_assembler; + +=head1 NAME + +CDNA::Genome_based_cdna_assembler + +=cut + +=head1 DESCRIPTION + +This module is used to assemble compatible cDNA alignments. The algorithm is as follows: +must describe this here. + +=cut + +use strict; +use CDNA::CDNA_alignment; +use Data::Dumper; +use base ("CDNA::Genome_based_cDNA_assembler"); + +=item new() + +=over 4 + +B instantiates a new cDNA assembler obj. + +B $sequence_sref + +$sequence_sref is a reference to a scalar containing the genomic sequence string. + +B $obj_href + +$obj_href is the object reference newly instantiated by this new method. + +=back + +=cut + +sub new () { + my $package_name = shift; + my $self = {}; + bless ($self, $package_name); + $self->_init(@_); + return ($self); +} + + +sub _init { + my $self = shift; + $self->CDNA::Genome_based_cDNA_assembler::_init(@_); + $self->{show_matrix} = 0; + $self->{compatibilities} = []; #retains results of compatibility comparisons: + $self->{incapsulations} = []; + $self->{LobjsForient} = []; #non-fli single-segment alignments forced in forward orientation. + $self->{LobjsRorient} = []; #non-fli single segment alignments forced in reverse orientation. + $self->{indices_included} = []; #remember what indices are built into assemblies. +} + + +=item assemble_alignments() + +=over 4 + +B assembles a series of cDNA aligmnments into one or more cDNA assemblies using a directed acyclic graph. + +B @alignments + +@alignments is an array of CDNA::CDNA_alignment objects + +B none. + +=back + +=cut + + +sub assemble_alignments { + my $self = shift; + my @alignments = @_; + @alignments = sort {$a->{lend}<=>$b->{lend}} @alignments; #keep in order of lend across genomic sequence to provide a layout. + $self->{incoming_alignments} = [@alignments]; + + my $num_alignments = $#alignments + 1; + + ## Initialize globals. + $self->{incapsulations} = []; #initialize. + $self->{compatibilities} = {}; #initialize. + $self->{indices_included} = []; #initialize + + $self->{assemblies} = []; #initialize + + #precompute compatible alignment pairs and full encapsulations. + $self->determine_compatibilities_and_encapsulations(); + + ###################################### + # Do Forward Scan Dynamic Programming. + my $LobjsForient = $self->{LobjsForient} = []; + my $LobjsRorient = $self->{LobjsRorient} = []; + $self->force_flexorient('+'); + $self->do_full_Fscan($LobjsForient); + $self->force_flexorient('-'); + $self->do_full_Fscan($LobjsRorient); + + &Describe_containment($LobjsForient, $LobjsRorient) if $SEE; + + ## Get the highest scoring Lobj + + my $struct = $self->get_top_alignment_indices(); + + ## Create the highest scoring assembly. + my $assembly = $self->create_assembly($struct); + my $num_cdnas_included = $struct->{num_cdnas}; + + if ( $num_cdnas_included == $num_alignments) { return();} + + + ####################################### + ## Do Reverse Scan Dynamic Programming (if cDNAs exist and not included in assembly above). + $self->force_flexorient('+'); + $self->do_full_Rscan($LobjsForient); + $self->force_flexorient('-'); + $self->do_full_Rscan($LobjsRorient); + ## Create assemblies for each cDNA not included yet. + my @asmbl_sets; + my %seen; + for (my $i = 0; $i < $num_alignments; $i++) { + unless ($self->{indices_included}->[$i]) { + print "Not Included: $i, Doing Back trace starting at $i.\n" if $SEE; + my $struct = $self->get_top_alignment_indices_starting_index($i); + push (@asmbl_sets, $struct); + } + } + @asmbl_sets = reverse sort {$a->{num_cdnas}<=>$b->{num_cdnas}} @asmbl_sets; + my $all_included_flag = 0; + while ((!$all_included_flag) && @asmbl_sets) { + my $asmbl_set = shift @asmbl_sets; + $all_included_flag = $self->test_and_add_asmbl($asmbl_set); + } + unless ($all_included_flag) { + die "FATAL: All assemblies not included!!\n"; + } + +} + +# private. +sub do_full_Fscan { + my $self = shift; + my $Lobjs = shift; + + print "\n# Doing full Fscan\n" if $SEE; + my $alignments_aref = $self->{incoming_alignments}; + my $num_alignments = $#{$alignments_aref} + 1; + + for (my $i = 0; $i < $num_alignments; $i++) { + my $Lobj = Lobject->new($i, $self); + if ($i != 0) { #first cDNA, base case. + ## Must compare to previous alignments: + my $top_score = 0; + my $top_scoring_index = -1; + my $i_alignment = $alignments_aref->[$i]; + for (my $j = $i - 1; $j >= 0; $j--) { + my $j_alignment = $alignments_aref->[$j]; + my $prevLobj = $Lobjs->[$j]; + print "Comparing $i to $j\n" if $SEE; + my $compatibility = $self->{compatibilities}->{$i}->{$j}; + my $j_contains_i = $prevLobj->{contained_cdna_indices}->{$i}; + my $i_contains_j = $Lobj->{contained_cdna_indices}->{$j}; + print "\tcompat: $compatibility\tj_contains_i: $j_contains_i\n" if $SEE; + if ($compatibility && (!($j_contains_i||$i_contains_j))) { + unless (&runtime_compatible($i_alignment, $j_alignment)) {next;} + my $curr_Lscore = $Lobjs->[$j]->{LscoreF}; + my $Cscore = &get_Cscore($Lobj, $prevLobj); + my $curr_total_score = $curr_Lscore + $Cscore; + + print "\tFSCAN score: ($i,$j) = $curr_total_score\n" if $SEE; + if ($curr_total_score > $top_score) { + $top_scoring_index = $j; + $top_score = $curr_total_score; + print "\t*making topscore.\n" if $SEE; + + } + } + } + if ($top_scoring_index > -1) { + my $topLobj = $Lobjs->[$top_scoring_index]; + $Lobj->{fromLptr} = $topLobj; + $Lobj->{LscoreF} = $top_score; + + } + + } + $Lobjs->[$i] = $Lobj; + } +} + +# private. +sub do_full_Rscan { + my $self = shift; + my $Lobjs = shift; + + + print "\n# Doing full Rscan\n" if $SEE; + my $alignments_aref = $self->{incoming_alignments}; + my $num_alignments = $#{$alignments_aref} + 1; + + for (my $i = $num_alignments - 2; $i >= 0; $i--) { + my $Lobj = $Lobjs->[$i]; + my $i_alignment = $alignments_aref->[$i]; + ## Must compare to previous alignments: + my $top_score = 0; + my $top_scoring_index = -1; + for (my $j = $i + 1; $j < $num_alignments; $j++) { + my $j_alignment = $alignments_aref->[$j]; + my $nextLobj = $Lobjs->[$j]; + print "Comparing $i to $j\n" if $SEE; + my $compatibility = $self->{compatibilities}->{$i}->{$j}; + my $j_contains_i = $nextLobj->{contained_cdna_indices}->{$i}; + my $i_contains_j = $Lobj->{contained_cdna_indices}->{$j}; + print "\tcompat: $compatibility\tj_contains_i: $j_contains_i\n" if $SEE; + if ($compatibility && (!($j_contains_i||$i_contains_j))) { + unless (&runtime_compatible($i_alignment, $j_alignment)) { next;} + my $curr_Lscore = $Lobjs->[$j]->{LscoreR}; + my $Cscore = &get_Cscore($Lobj, $nextLobj); + my $curr_total_score = $curr_Lscore + $Cscore; + + print "\tRSCAN score: ($i,$j) = $curr_total_score\n" if $SEE; + if ($curr_total_score > $top_score) { + $top_scoring_index = $j; + $top_score = $curr_total_score; + print "\t*making topscore.\n" if $SEE; + + } + } + } + if ($top_scoring_index > -1) { + my $topLobj = $Lobjs->[$top_scoring_index]; + $Lobj->{toLptr} = $topLobj; + $Lobj->{LscoreR} = $top_score; + } + + } +} + + +# private. +sub get_Cscore { + ## Cscore is the number of cdnas contained in a but not in b including a itself + my ($a, $b) = @_; + my $score = 0; + my $a_container_ref = $a->{contained_cdna_indices}; + my $b_container_ref = $b->{contained_cdna_indices}; + foreach my $a_index (keys %$a_container_ref) { + unless ($b_container_ref->{$a_index}) { + $score++; + } + } + return ($score); +} + + +=over 4 + +=item encapsulates() + +B Returns true (1) if alignment A encapsulates the span of alignment B + +B ($alignment_A, $alignment_B) + +Alignments A and B are of type CDNA::CDNA_alignment + +B [1|0] + +=back + +=cut + + +sub encapsulates { + my $self = shift; + my ($alignmentA, $alignmentB) = @_; + my ($alend, $arend) = $alignmentA->get_coords(); + my ($blend, $brend) = $alignmentB->get_coords(); + + if ($blend >= $alend && $brend <= $arend) { + return (1); + } else { + return (0); + } +} + + +sub determine_compatibilities_and_encapsulations { + my $self = shift; + + my $alignments_aref = $self->{incoming_alignments}; + my $num_alignments = $#{$alignments_aref} + 1; + + + #initialize incapsulations list + for (my $i = 0; $i < $num_alignments; $i++) { + $self->{incapsulations}->[$i] = []; + } + + #compare each downstream alignment to the upstream alignment for compatiblity and encapsulation: + for (my $i = 0; $i < $num_alignments; $i++) { + for (my $j = $i + 1; $j < $num_alignments; $j++) { + + my $alignment_i = $alignments_aref->[$i]; + my $alignment_j = $alignments_aref->[$j]; + + + my $can_merge = 0; + if ($self->can_merge($alignment_i, $alignment_j)) { + $can_merge = 1; + # set compatibility flags. + $self->{compatibilities}->{$j}->{$i} = 1; + $self->{compatibilities}->{$i}->{$j} = 1; + } + + my $align_i_single_seg = ($alignment_i->get_num_segments() == 1) ? 1:0; + my $align_j_single_seg = ($alignment_j->get_num_segments() == 1) ? 1:0; + + + ## Analyze compatible alignments for containment. Single segment alignments may conflict if in opposite orientations and fli-status, so should check these incompatible alignments for containment as well to avoid multiple identical assemblies from being generated. + + if ($can_merge || ($align_i_single_seg && $align_j_single_seg)) { + + ## Check for encapsulation: + if ($self->encapsulates($alignment_j, $alignment_i)) { + push (@{$self->{incapsulations}->[$j]}, $i); + print "$j contains $i\n" if $SEE; + } + if ($self->encapsulates($alignment_i, $alignment_j)) { + push (@{$self->{incapsulations}->[$i]}, $j); + print "$i contains $j\n" if $SEE; + } + + } + } + } +} + +sub force_flexorient { + my $self = shift; + my $orient = shift; + my $alignments_aref = $self->{incoming_alignments}; + my $num_alignments = $#{$alignments_aref} + 1; + ## Fix orientations of fli and multi-segment alignments: + for (my $i = 0; $i < $num_alignments; $i++) { + my $alignment = $alignments_aref->[$i]; + my $num_segments = $alignment->get_num_segments(); + my $spliced_orientation = $alignment->get_spliced_orientation(); + + if ($spliced_orientation =~ /^[+-]$/) { # set specifically + $alignment->{fixed_orient} = $spliced_orientation; ## adding tag, using only in this module. + } else { + print "Setting $i to flex orient: $orient.\n" if $SEE; + $alignment->{fixed_orient} = $orient; + } + } + +} + + +sub back_trace { + my $self = shift; + my $top_scoring_index = shift; + my $Lobjs = shift; + + print "Back_trace: " if $SEE; + my $Lobj = $Lobjs->[$top_scoring_index]; + ## Traverse the fromLptrs + + my %unique_indices; + my @trace_indices; + while ($Lobj != 0) { + my $index = $Lobj->{myIndex}; + print "$index " if $SEE; + push (@trace_indices, $index); + # get composed accs: + foreach my $contained_index ($Lobj->get_contained_indices()) { + $unique_indices{$contained_index} = 1; + } + $Lobj = $Lobj->{fromLptr}; + } + print "\n" if $SEE; + my @unique = keys %unique_indices; + my $num_cdnas_included = $#unique + 1; + my $struct = {trace_indices=>\@trace_indices, + num_cdnas=>$num_cdnas_included, + all_indices=>\@unique}; + return ($struct); +} + + +sub forward_trace { + my $self = shift; + my $index = shift; + my $Lobjs = shift; + + print "Forward_trace: " if $SEE; + my $Lobj = $Lobjs->[$index]; + ## Traverse the fromLptrs + my @trace_indices; + my %unique_indices; + while ($Lobj != 0) { + my $index = $Lobj->{myIndex}; + print " $index" if $SEE; + push (@trace_indices, $index); + # get composed accs: + foreach my $contained_index ($Lobj->get_contained_indices()) { + $unique_indices{$contained_index} = 1; + } + $Lobj = $Lobj->{toLptr}; + } + + + print "\n" if $SEE; + my @unique = keys %unique_indices; + my $num_cdnas_included = $#unique + 1; + my $struct = {trace_indices=>\@trace_indices, + num_cdnas=>$num_cdnas_included, + all_indices=>\@unique}; + + return ($struct); +} + + +sub create_assembly { + my $self = shift; + my $struct = shift; + my $indices_aref = $struct->{trace_indices}; + my $alignments_aref = $self->{incoming_alignments}; + my @indices = sort {$a<=>$b} @$indices_aref; + my $all_indices_ref = $struct->{all_indices}; + my @all_indices = sort {$a<=>$b} @$all_indices_ref; + print "Creating assembly: @indices\n" if $SEE; + + my $first_index = shift @indices; + my $alignment = $alignments_aref->[$first_index]; + $self->{indices_included}->[$first_index] = 1; + print "Index: $first_index\t" . $alignment->toToken() . "\n" if $SEE; + my $assembly = $alignment->clone(); + + my @alignments_to_assemble; + # first index already included in assembly; ...now, include the others. + foreach my $index (@indices) { + my $alignment = $alignments_aref->[$index]; + print "Index: $index\t" . $alignment->toToken() . "\n" if $SEE; + $assembly = $self->merge_alignments($assembly, $alignment); + } + print "assembly contains: @all_indices\n" if $SEE; + my @all_accs; + + foreach my $index (@all_indices) { + $self->{indices_included}->[$index] = 1; + my $acc = $alignments_aref->[$index]->get_acc(); + push (@all_accs, $acc); + } + my $acc = $self->merge_accs(@all_accs); + $assembly->set_acc($acc); + + push (@{$self->{assemblies}}, $assembly);; + + return ($assembly); +} + +sub runtime_compatible { + my ($a, $b) = @_; + my $a_fixed_orient = $a->{fixed_orient}; + my $b_fixed_orient = $b->{fixed_orient}; + print "a($a_fixed_orient) vs. b($b_fixed_orient)\n" if $SEE; + if ($a_fixed_orient eq $b_fixed_orient) { + print "runtime compatible.\n" if $SEE; + return (1); + } else { + print "NOT runtime compatible\n" if $SEE; + return (0); + } +} + + +sub get_top_alignment_indices { + my $self = shift; + my $LobjsForient = $self->{LobjsForient}; + my $LobjsRorient = $self->{LobjsRorient}; + + ## Process Forient (singletons forced Forward orient) + my $top_Forient_index = $self->top_scoring_index_from_Lobjs($LobjsForient, 'LscoreF'); + my $ForientStruct = $self->back_trace($top_Forient_index, $LobjsForient); + + ## Process Rorient (singletons forced Reverse orient) + my $top_Rorient_index = $self->top_scoring_index_from_Lobjs($LobjsRorient, 'LscoreF'); + my $RorientStruct = $self->back_trace($top_Rorient_index, $LobjsRorient); + if ($ForientStruct->{num_cdnas} >= $RorientStruct->{num_cdnas}) { + return ($ForientStruct); + } else { + return ($RorientStruct); + } +} + +sub get_top_alignment_indices_starting_index { + my $self = shift; + my $index = shift; + my $LobjsForient = $self->{LobjsForient}; + my $LobjsRorient = $self->{LobjsRorient}; + my $LobjsForient_num_contained = $LobjsForient->[$index]->get_contained_indices(); + my $LobjsRorient_num_contained = $LobjsRorient->[$index]->get_contained_indices(); + + ## Process Forient (singletons forced forward orient) + my $ForientBtraceStruct = $self->back_trace($index, $LobjsForient); + my $ForientFtraceStruct = $self->forward_trace($index, $LobjsForient); + my $ForientNumCdnas = $ForientBtraceStruct->{num_cdnas} + $ForientFtraceStruct->{num_cdnas} -$LobjsForient_num_contained; + + ## Process Rorient (singletons forced reverse orient) + my $RorientBtraceStruct = $self->back_trace($index, $LobjsRorient); + my $RorientFtraceStruct = $self->forward_trace($index, $LobjsRorient); + my $RorientNumCdnas = $RorientBtraceStruct->{num_cdnas} + $RorientFtraceStruct->{num_cdnas} - $LobjsRorient_num_contained; + + my ($Ftrace, $Btrace,$total_cdnas); + if ($ForientNumCdnas > $RorientNumCdnas) { + ($Ftrace, $Btrace) = ($ForientFtraceStruct, $ForientBtraceStruct); + $total_cdnas = $ForientNumCdnas; + } else { + ($Ftrace, $Btrace) = ($RorientFtraceStruct, $RorientBtraceStruct); + $total_cdnas = $RorientNumCdnas; + } + + ## Create new struct + my @unique_trace = &unique_entries(@{$Ftrace->{trace_indices}}, @{$Btrace->{trace_indices}}); + my @all_indices = &unique_entries (@{$Ftrace->{all_indices}}, @{$Btrace->{all_indices}}); + if ($#all_indices +1 != $total_cdnas) { + die "Error: total number of alignments not calculated correctly.\n"; + } + print "FnB trace from index[$index] yields assembly containing indices [@all_indices]\n" if $SEE; + my $struct = {trace_indices=>\@unique_trace, + num_cdnas=>$total_cdnas, + all_indices=>\@all_indices}; + return ($struct); +} + + + + +sub unique_entries { + my @x = @_; + my %z; + foreach my $y (@x) { + $z{$y}=1; + } + return (keys %z); +} + + +sub top_scoring_index_from_Lobjs { + my $self = shift; + my $Lobjs = shift; + my $score_type = shift; + + my $alignments_aref = $self->{incoming_alignments}; + my $num_alignments = $#{$alignments_aref} + 1; + my $top_scoring_index = -1; + my $top_score = -1; + + for (my $i = 0; $i < $num_alignments; $i++) { + my $Lscore = $Lobjs->[$i]->{$score_type}; + if ($Lscore > $top_score) { + $top_scoring_index = $i; + $top_score = $Lscore; + } + } + return ($top_scoring_index); +} + + +sub test_and_add_asmbl { + my $self = shift; + my $asmbl_set = shift; + my $all_indices_ref = $asmbl_set->{all_indices}; + my $indices_included_ref = $self->{indices_included}; +## If the asmbl_set contains a cDNA missing from an assembly inclusion, then add it. + + + my $num_alignments = $#{$self->{incoming_alignments}} + 1; + my $contains_unincorporated = 0; + foreach my $index (@$all_indices_ref) { + if (! $indices_included_ref->[$index]) { + $contains_unincorporated = 1; + last; + } + } + if ($contains_unincorporated) { + $self->create_assembly($asmbl_set); + } + + ## Test to see if all cDNAs are accounted for in assemblies. + my $all_included_flag = 1; + + for (my $i =0; $i < $num_alignments; $i++) { + if (! $indices_included_ref->[$i]) { + print "$i not included!\n" if $SEE; + $all_included_flag = 0; + } + } + return ($all_included_flag); +} + +sub Describe_containment { + my ($LobjsForient, $LobjsRorient) = @_; + for (my $i = 0; $i <= $#$LobjsForient; $i++) { + print "Fixed Forient: index $i contains: " . join (" ", $LobjsForient->[$i]->get_contained_indices()) . "\n"; + print "Fixed Rorient: index $i contains: " . join (" ", $LobjsRorient->[$i]->get_contained_indices()) . "\n"; + } + print "Lscores:\n"; + for (my $i = 0; $i <= $#$LobjsForient; $i++) { + print "$i: F-forced singleton orient Lscores: F: " . $LobjsForient->[$i]->{LscoreF} . " R: " . $LobjsForient->[$i]->{LscoreR} . "\n"; + print "$i: R-forced singleton orient Lscores: F: " . $LobjsRorient->[$i]->{LscoreF} . " R: " . $LobjsRorient->[$i]->{LscoreR} . "\n"; + } +} + + + + + +####################### +package Lobject; + +sub new { + my $packagename = shift; + my $index = shift; + my $assembler_ref = shift; + my $self = { + contained_cdna_indices =>{ $index => 1 #include as containing itself. + }, + myIndex => $index, #remember Lobj position. + LscoreF => 1, #Forward scan Lscore. + LscoreR => 1, #Reverse scan Lscore. + toLPtr => 0, # used in Rscan for forward tracking. + fromLPtr => 0 # used in Fscan for backtracking + }; + + ## populate list of contained entries. + my @contained_indices = @{$assembler_ref->{incapsulations}->[$index]}; + foreach my $contained_index (@contained_indices) { + $self->{contained_cdna_indices}->{$contained_index} = 1; + $self->{LscoreF}++; + $self->{LscoreR}++; + } + + bless ($self, $packagename); + return ($self); +} + +sub get_contained_indices { + my $self = shift; + return (keys %{$self->{contained_cdna_indices}}); +} + + +1; #EOM + + + + + diff --git a/99.scripts/trinity_utils/PerlLib/CDNA/Overlap_assembler.pm b/99.scripts/trinity_utils/PerlLib/CDNA/Overlap_assembler.pm new file mode 100644 index 0000000..c6f7dd2 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/CDNA/Overlap_assembler.pm @@ -0,0 +1,119 @@ +package main; +our $SEE = 0; + + +package CDNA::Overlap_assembler; + +use strict; + +sub new { + my $packagename = shift; + my $self = { + node_list => [] + }; + bless ($self, $packagename); + return ($self); +} + +sub add_cDNA { + my $self = shift; + my ($cdna_acc, $end5, $end3) = @_; + my ($cdna_start, $cdna_stop) = sort {$a<=>$b} ($end5, $end3); + my $node = Cdna_node->new($cdna_acc, $cdna_start, $cdna_stop); + push (@{$self->{node_list}}, $node); +} + + +#### +sub build_clusters { + my $self = shift; + my $node_list_aref = $self->{node_list}; + @{$node_list_aref} = sort {$a->{lend}<=>$b->{lend}} @{$node_list_aref}; #sort by lend coord. + ## set indices + for (my $i = 0; $i <= $#{$node_list_aref}; $i++) { + $node_list_aref->[$i]->{myIndex} = $i; + } + + my @clusters; + my $first_node = $node_list_aref->[0]; + my $start_pos = 0; + my ($exp_left, $exp_right) = ($first_node->{lend}, $first_node->{rend}); + print $first_node->{acc} . " ($exp_left, $exp_right)\n" if $SEE; + for (my $i = 1; $i <= $#{$node_list_aref}; $i++) { + my $curr_node = $node_list_aref->[$i]; + my ($lend, $rend) = ($curr_node->{lend}, $curr_node->{rend}); + print $curr_node->{acc} . " ($lend, $rend)\n" if $SEE; + if ($exp_left <= $rend && $exp_right >= $lend) { #overlap + $exp_left = &min($exp_left, $lend); + $exp_right = &max($exp_right, $rend); + print "overlap. New expanded coords: ($exp_left, $exp_right)\n" if $SEE; + } else { + print "No overlap; Creating cluster: " if $SEE; + my @cluster; + for (my $j=$start_pos; $j < $i; $j++) { + my $acc = $node_list_aref->[$j]->{acc}; + push (@cluster, $acc); + print "$acc, " if $SEE; + } + push (@clusters, [@cluster]); + $start_pos = $i; + ($exp_left, $exp_right) = ($lend, $rend); + print "\nResetting expanded coords: ($lend, $rend)\n" if $SEE; + } + } + + print "# Adding final cluster.\n" if $SEE; + if ($start_pos != $#{$node_list_aref}) { + print "final cluster: " if $SEE; + my @cluster; + for (my $j = $start_pos; $j <= $#{$node_list_aref}; $j++) { + my $acc = $node_list_aref->[$j]->{acc}; + print "$acc, " if $SEE; + push (@cluster, $acc); + } + push (@clusters, [@cluster]); + print "\n" if $SEE; + } else { + my $acc = $node_list_aref->[$start_pos]->{acc}; + push (@clusters, [$acc]); + print "adding final $acc.\n" if $SEE; + } + return (@clusters); +} + +sub min { + my (@x) = @_; + @x = sort {$a<=>$b} @x; + my $min = shift @x; + return ($min); +} + +sub max { + my @x = @_; + @x = sort {$a<=>$b} @x; + my $max = pop @x; + return ($max); +} + +################################################################# +package Cdna_node; +use strict; + +sub new { + my $packagename = shift; + my ($acc, $lend, $rend) = @_; + my $self = { acc=>$acc, + lend=>$lend, + rend=>$rend, + myIndex=>undef(), + overlapping_indices=>[] + }; + bless ($self, $packagename); + return ($self); +} + +1; #EOM + + + + diff --git a/99.scripts/trinity_utils/PerlLib/CDNA/PASA_alignment_assembler.pm b/99.scripts/trinity_utils/PerlLib/CDNA/PASA_alignment_assembler.pm new file mode 100644 index 0000000..fcf9dab --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/CDNA/PASA_alignment_assembler.pm @@ -0,0 +1,437 @@ +#!/usr/local/bin/perl + +package main; +our $SEE; + + +package CDNA::PASA_alignment_assembler; + +=head1 NAME + +CDNA::PASA_alignment_assembler + +=cut + +=head1 DESCRIPTION + +This module is used to assemble compatible cDNA alignments. The algorithm is as follows: +must describe this here. + +=cut + +use strict; +use CDNA::CDNA_alignment; +use Data::Dumper; +use Carp; + +## File scoped globals: +my $DELIMETER = "$;,"; +our $FUZZLENGTH = 20; + + +=item new() + +=over 4 + +B instantiates a new cDNA assembler obj. + +B none + +B $obj_href + +$obj_href is the object reference newly instantiated by this new method. + +=back + +=cut + +sub new { + my $package_name = shift; + my $self = {}; + bless ($self, $package_name); + $self->_init(@_); + return ($self); +} + + +sub _init { + my $self = shift; + $self->{incoming_alignments} = []; #these are the alignments to be assembled. + $self->{assemblies} = []; #contains list of all singletons and assemblies. + $self->{fuzzlength} = $FUZZLENGTH; #default setting. + + my $pasa_bin = `sh -c "command -v pasa"`; + $pasa_bin =~ s/\s//g; + + unless (-x $pasa_bin) { + confess "Error, pasa binary [$pasa_bin] isn't executable or couldn't be found."; + } + + $self->{pasa_bin} = $pasa_bin; + +} + + +=item assemble_alignments() + +=over 4 + +B assembles a series of cDNA aligmnments into one or more cDNA assemblies using a directed acyclic graph. + +B @alignments + +@alignments is an array of CDNA::CDNA_alignment objects + +B none. + +=back + +=cut + + +sub assemble_alignments { + my $self = shift; + my @alignments = @_; + @alignments = sort {$a->{lend}<=>$b->{lend}} @alignments; #keep in order of lend across genomic sequence to provide a layout. + $self->{incoming_alignments} = [@alignments]; + my %accs; + my %spliced_orientations; # track so can set later in each assembly based on content. + my %aligned_orientations; + my %FL_accs; + + foreach my $alignment (@alignments) { + my $acc = $alignment->get_acc(); + $accs{$acc} = 0; + my $spliced_orient = $alignment->get_spliced_orientation(); + $spliced_orientations{$acc} = $spliced_orient; + my $aligned_orient = $alignment->get_orientation(); + $aligned_orientations{$acc} = $aligned_orient; + $FL_accs{$acc} = $alignment->is_fli(); + } + $self->{accs_in_assemblies} = \%accs; + + my $num_alignments = $#alignments + 1; + + $self->force_flexorient('+'); + my @assemblies = $self->pasa_cpp_assemblies('+'); + $self->force_flexorient('-'); + push (@assemblies, $self->pasa_cpp_assemblies('-')); + + # sort in order of decreasing score + @assemblies = reverse sort {$a->{num_contained_aligns}<=>$b->{num_contained_aligns}} @assemblies; + + if ($SEE) { + print "\n\nScore summary for all assemblies (nr set unchosen):\n\n"; + foreach my $assembly (@assemblies) { + print "score: " . $assembly->{num_contained_aligns} . ", " . $assembly->toToken . "\n"; + } + print "\n\n"; + } + my @report_assemblies; + my $still_missing = 1; + foreach my $assembly (@assemblies) { + print "\nAnalyzing assembly.\n" if $SEE; + my $contained_aligns_aref = $assembly->{contained_aligns}; + my $have_unseen = 0; + my $spliced_orient = '?'; + my $is_fli = 0; + my %aligned_orient_counts; + + foreach my $acc (@$contained_aligns_aref) { + print "got: $acc\n" if $SEE; + # check to see if we've encountered this one yet. + unless ($accs{$acc}) { + $have_unseen = 1; + } + $accs{$acc} = 1; + my $curr_spliced_orient = $spliced_orientations{$acc}; + print "curr_spliced_orient: $curr_spliced_orient\n" if $SEE; + if ($curr_spliced_orient ne '?') { + if ($spliced_orient ne '?' && $spliced_orient ne $curr_spliced_orient) { + ## cannot have conflicting spliced orientations in the assembly: corruption. + die "Fatal: conflicting spliced orientations in current PASA assembly results (spliced_orient: $spliced_orient, $acc has $curr_spliced_orient).\n"; + } + $spliced_orient = $curr_spliced_orient; ## retain original spliced orientation. + } + if ( (!$is_fli) && $FL_accs{$acc}) { + $is_fli = 1; + } + ## track aligned orientation + $aligned_orient_counts{ $aligned_orientations{$acc} } ++; + + } + + ## set assembly orientation + $assembly->set_spliced_orientation($spliced_orient); + if ($spliced_orient eq '?') { + ## set aligned orientation based on a majority vote + my @orients = reverse sort { $aligned_orient_counts{$a} <=> $aligned_orient_counts{$b} } keys %aligned_orient_counts; + my $winning_aligned_orient = shift @orients; + $assembly->set_orientation($winning_aligned_orient); + } + else { + # got spliced orientation, use it for aligned orientation too. + $assembly->set_orientation($spliced_orient); + } + + $assembly->set_fli_status($is_fli); + + if ($have_unseen) { + print $assembly->toToken . "\n" if $SEE; + push (@report_assemblies, $assembly); + } + $still_missing = 0; + foreach my $key (keys %accs) { + if (! $accs{$key}) { + $still_missing = 1; + print "still missing: $key\n" if $SEE; + } + } + if (! $still_missing) { + last; #got them all. + } + } + + $self->{assemblies} = \@report_assemblies; + + if ($still_missing) { + die "Didn't obtain assemblies describing all maximal assemblies.\n"; + } + + +} + + +sub pasa_cpp_assemblies { + my $self = shift; + my $forced_orient = shift; + + my $prev_input_sep = $/; + $/ = "\n"; + + my $sequence_ref; + my $incoming_alignments_aref = $self->{incoming_alignments}; + # create input file for pasa-cpp implementation: + my $pasa_input = "pasa.$$.$forced_orient.in"; + my $pasa_output = "pasa.$$.$forced_orient.out"; + my @assemblies; + + open (TMPIN, ">$pasa_input") or die "Can't open file $pasa_input"; + foreach my $alignment (@$incoming_alignments_aref) { + my $acc = $alignment->get_acc(); + ## commas not allowed in acc name: + if ($acc =~ /,/) { + die "ERROR, $acc accession contains comma(s). This is not allowed.\n"; + } + my $orient = $alignment->{fixed_orient}; + my $alignText = "$acc,$orient"; + unless (ref $sequence_ref) { + $sequence_ref = $alignment->get_genomic_seq_ref(); + } + foreach my $seg ($alignment->get_alignment_segments()) { + my ($lend, $rend) = $seg->get_coords(); + $alignText .= ",$lend-$rend"; + } + print TMPIN $alignText . "\n"; + } + close TMPIN; + + if ($SEE) { + print "PASA_INPUT ($forced_orient):\n====\n"; + system "cat $pasa_input"; + print "====\n"; + } + my $cmd = $self->{pasa_bin} . " $pasa_input > $pasa_output"; + my $ret = system $cmd; + if ($ret) { + system "mv $pasa_input pasa_killer.input"; + print STDERR "PASA died on input file. See pasa_killer.input"; + die; + } else { + + # process the output. + open (TMPOUT, $pasa_output) or die "Can't open $pasa_output"; + while () { + if (/assembly:\s\(\d+\)\scontains\salignments:\s\[([^\]]+)\]\swith\sstructure\s\[([^\]]+)\]/) { + print "Extracting assembly output: $_" if $SEE; + my $acclist = $1; + my $aligndescript = $2; + my @x = split (/,/, $aligndescript); + + shift @x; + my $orient = shift @x; + my @alignSegs; + my $length = 0; + foreach my $coordset (@x) { + my ($lend, $rend) = sort {$a<=>$b} split (/-/, $coordset); + my $seg = new CDNA::Alignment_segment($lend, $rend); + $length += ($rend - $lend) + 1; + push (@alignSegs, $seg); + } + my $assembly = new CDNA::CDNA_alignment($length, \@alignSegs, $sequence_ref); + + my @accs = split (/,/, $acclist); + my $num_accs = $#accs + 1; + $assembly->{contained_aligns} = [@accs]; + $assembly->{num_contained_aligns} = $num_accs; + + $acclist =~ s/,/\//g; #convert list of accessions into a new accession representing a single entry (unity) + # if we keep the commas, use of this assembly in future PASA runs will break the assembler + # because of the input file requirements. + $assembly->set_acc($acclist); + + push (@assemblies, $assembly); + } + + } + close TMPOUT; + + if ($SEE) { + print "PASA_OUTPUT ($forced_orient):\n####\n"; + system "cat $pasa_output"; + print "####\n"; + } + unlink ($pasa_input, $pasa_output) unless $SEE; + + } + + $/ = $prev_input_sep; ## restore + + return (@assemblies); +} + + +sub force_flexorient { + my $self = shift; + my $orient = shift; + my $alignments_aref = $self->{incoming_alignments}; + my $num_alignments = $#{$alignments_aref} + 1; + ## Fix orientations of fli and multi-segment alignments: + for (my $i = 0; $i < $num_alignments; $i++) { + my $alignment = $alignments_aref->[$i]; + my $num_segments = $alignment->get_num_segments(); + my $spliced_orientation = $alignment->get_spliced_orientation(); + + if ($spliced_orientation =~ /^[+-]$/) { # set specifically + $alignment->{fixed_orient} = $spliced_orientation; ## adding tag, using only in this module. + } else { + print "Setting $i to flex orient: $orient.\n" if $SEE; + $alignment->{fixed_orient} = $orient; + } + } + +} + + +sub unique_entries { + my @x = @_; + my %z; + foreach my $y (@x) { + $z{$y}=1; + } + return (keys %z); +} + + +=item get_assemblies() + +=over 4 + +B returns all the alignment assemblies resulting from the assembly procedure. + +B none. + +B @assemblies + +@assemblies is an array of CDNA::CDNA_alignment objects. + +use the get_acc() method of the alignment object to retrieve all the accessions of the cDNAs that were merged into the assembly. + +=back + +=cut + + +sub get_assemblies { + my $self = shift; + return (@{$self->{assemblies}}); +} + +=item toAlignIllustration() + +=over 4 + +B illustrates the individual cDNAs to be assembled along with the final products. + +B $max_line_chars(optional) + +$max_line_chars is an integer representing the maximum number of characters in a single line of output to the terminal. The default is 100. + +B $alignment_illustration_text + +$alignment_illustration_text is a string containing a paragraph of text which illustrates the alignments and assemblies. An example is below: + + ---> <--> <-----> <---> <---------------- (+)gi|1199466 + + ---> <--> <-----> <---> <------------ (+)gi|1209702 + +----> <--> <---- (+)AV827070 + +----> <--> <--- (+)AV828861 + +----> <--> <--- (+)AV830936 + + ---> <--> <- (+)H36350 + +ASSEMBLIES: (1) + +----> <--> <-----> <---> <---------------- (+) gi|1199466, gi|1209702, AV827070, AV828861, AV830936, H36350 + + + + +=back + +=cut + + ; + +sub toAlignIllustration () { + my $self = shift; + my $max_line_chars = shift; + $max_line_chars = ($max_line_chars) ? $max_line_chars : 100; #if not specified, 100 chars / line is default. + + ## Get minimum coord for relative positioning. + my @coords; + my @alignments = @{$self->{incoming_alignments}}; + foreach my $alignment (@alignments) { + my @c = $alignment->get_coords(); + push (@coords, @c); + } + @coords = sort {$a<=>$b} @coords; + print "coords: @coords\n" if $::SEE; + my $min_coord = shift @coords; + my $max_coord = pop @coords; + my $rel_max = $max_coord - $min_coord; + my $alignment_text = ""; + ## print each alignment followed by assemblies: + my $num_alignments = $#alignments + 1; + $alignment_text .= "Individual Alignments: ($num_alignments)\n"; + my $i = 0; + foreach my $alignment (@alignments) { + $alignment_text .= (sprintf ("%3d ", $i)) . $alignment->toAlignIllustration($min_coord, $rel_max, $max_line_chars) . "\n"; + $i++; + } + + my @assemblies = @{$self->{assemblies}}; + my $num_assemblies = $#assemblies + 1; + $alignment_text .= "\n\nASSEMBLIES: ($num_assemblies)\n"; + foreach my $assembly (@assemblies) { + $alignment_text .= " " . $assembly->toAlignIllustration($min_coord, $rel_max, $max_line_chars) . "\n"; + } + + return ($alignment_text); +} + + +1; diff --git a/99.scripts/trinity_utils/PerlLib/CDNA/Splice_graph_assembler.pm b/99.scripts/trinity_utils/PerlLib/CDNA/Splice_graph_assembler.pm new file mode 100644 index 0000000..32f62fe --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/CDNA/Splice_graph_assembler.pm @@ -0,0 +1,2425 @@ +#!/usr/local/bin/perl + +package main; +our $SEE; + +package CDNA::Splice_graph_assembler; + +use strict; +use warnings; +use Carp; +use Overlap_piler; +use CDNA::CDNA_alignment; +use CDNA::Alignment_segment; +use Data::Dumper; +no warnings "recursion"; + +our $FUZZ_DIST = 20; # bp's not to trust at non-splice termini; untrustworthy alignment extensions. + +## allowable connections between gene structure components: +my %ACCEPTABLE_CONNECTIONS = ( "terminal_left_exon" => {"intron" => 1}, + "internal_exon" => {"intron" => 1}, + + "intron" => {"internal_exon" => 1, + "terminal_right_exon" => 1, + }, + ); + + +#### +sub new { + my $packagename = shift; + + my $self = { + _graph_nodes => [], # each and every graph node (main repository) + _incoming_alignments => [], # alignments to assemble + _assemblies => [], # resulting alignment assemblies + _unincorporated_alignments => [], + + _valid_splice_paths => [], # Splice_graph_path objects + _assembled_splice_paths => [], # compatible splice paths chained into splice path assemblies. + + ## other helpers + _graph_node_hashkey_lookup => {}, # key is lend,rend,type,orient + _graph_node_via_nodeID => {}, # node access by nodeID + + ## various node types: (all subsets of _graph_nodes and accessible via helpers above. + _internal_exons => [], + _introns => [], + _terminal_left_exons => [], + _terminal_right_exons => [], + _singleton_exons => [], + + ## misc + _terminal_exon_nodeID_to_nonsplice_position_list => {}, # track coords of non-splice site ends in terminal exon supports + }; + + bless ($self, $packagename); + + return ($self); +} + +#### +sub assemble_alignments { + my $self = shift; + my @alignments = @_; + + ## build the splicing graph. + $self->build_splicing_graph(@alignments); + + ## Chain together splice paths that are compatible: + print "-Chaining compatible splice paths\n" if $SEE; + $self->_chain_compatible_splice_paths(); + + ## Extend splice paths into maximally scoring complete paths with terminal exons + print "-Extending splice paths into maximal structures.\n" if $SEE; + $self->_extend_splice_paths_to_termini(); ## products are included in the assemblies list. + + ## Add the singleton assemblies + $self->_append_singletons_to_assembly_list(); ## convert the singletons to cdna alignments and add to assembly list + + ## assign the incoming alignments to the assemblies that contain them. + print "-Assigning transcript alignments to assemblies\n" if $SEE; + my $have_unincorporated_alignments_flag = $self->_correlate_assemblies_with_incoming_alignments(); + if ($have_unincorporated_alignments_flag) { + ## find paths from unincorporated terminal segments. + print "-Not all alignments were included. Exploring assemblies from uninorporated alignments\n" if $SEE; + $self->_explore_assemblies_from_unincorporated_alignments(); + print "-Assigning transcript alignments to assemblies, again...\n" if $SEE; + $have_unincorporated_alignments_flag = $self->_correlate_assemblies_with_incoming_alignments(); + if ($have_unincorporated_alignments_flag) { + ## something horribly has gone wrong. + confess "Error, not all incoming alignments are accounted for!\n" . $self->toString(); + } + } + + print $self->toString() if $SEE; +} + + +#### +sub build_splicing_graph { + my $self = shift; + my @alignments = @_; + + $self->set_incoming_alignments(@alignments); + + ## going to process the alignments in several phases: + # -decompose alignments into structural components + # -identify valid paths to connect components + # -extract assemblies using valid paths + # -associate transcripts with assemblies + + ## first, examine the alignments with the most segments first: + @alignments = reverse sort {$a->get_num_segments() <=> $b->get_num_segments()} @alignments; + + + ## build components of the splice graph + ## add internal exons and all introns to splice graph: (unambiguous structures) + ## terminal structures are not as well defined. Apply these later. + + print "-Decomposing alignments into gene structure components.\n" if $SEE; + foreach my $alignment (@alignments) { + if ($alignment->get_num_segments() > 1) { + $self->_decompose_unambiguous_alignment_structures_add_nodes($alignment); + } + } + + ## Add terminal exons: + ## Only add terminal exons where it's clear that it's not just part of an existing internal exon + + print "-Adding terminal exon segments\n" if $SEE; + my @single_segments; # capture those alignments w/o introns + + foreach my $alignment (@alignments) { + my $segment_orient = $alignment->get_spliced_orientation(); + if ($alignment->get_num_segments() > 1) { + foreach my $segment ($alignment->get_alignment_segments()) { + if ($segment->is_first() || $segment->is_last()) { + + my $exon_type = ($segment->is_first()) ? "terminal_left_exon" : "terminal_right_exon"; + + $self->_try_terminal_exon_addition($exon_type, $segment, $segment_orient); + + ## at this point, introns and internal exons are scored only by perfect support (exact boundaries). + ## terminal exons, on the other hand, have evidence scored based on boundary match. + ## Now, need to augment scores of internal exons that encompass terminal exons + + + my ($segment_lend, $segment_rend) = $segment->get_coords(); + + $self->_augment_internal_exon_scores_using_terminal_alignment_segment($exon_type, $segment_lend, + $segment_rend, $segment_orient); + } + } + } + else { + push (@single_segments, $alignment); + } + } + + ## reconstruct some internal exons based on overlapping right/left terminal exons + print "-Merging overlapping right-to-left terminal exons\n" if $SEE; + $self->_merge_overlapping_right_to_left_terminal_exons(); + + ## apply intron-less segments as evidence to internal and terminal segments + ## and instantiate single exons where they're not simply supporting existing structures. + print "-Analyzing intronless alignments\n" if $SEE; + $self->_analyze_intronless_alignments(@single_segments); + + print "-Building the splice graph\n" if $SEE; + $self->_build_splice_graph(); # could/should have done this earlier during alignment parse for efficiency. + + return; + +} + +#### +sub _find_internal_exons_encompassing_coords_and_share_boundary { + my $self = shift; + my ($exon_type, $segment_lend, $segment_rend, $segment_orient) = @_; + + unless ($exon_type =~ /left|right/) { + confess "invalide terminal exon type: $exon_type\n"; + } + + my @internal_exons_found; + + foreach my $internal_exon (@{$self->{_internal_exons}}) { + my $internal_exon_orient = $internal_exon->get_orient(); + my ($internal_exon_lend, $internal_exon_rend) = $internal_exon->get_coords(); + + ## if internal exon encompasses the terminal exon, add it to its evidence collection + if ( ($segment_orient eq $internal_exon_orient) + && + ($internal_exon_lend <= ($segment_lend + $FUZZ_DIST) && ($segment_rend - $FUZZ_DIST) <= $internal_exon_rend) # internal encompasses it + && + ( ($exon_type eq "terminal_right_exon" && $internal_exon_lend == $segment_lend) + || + ($exon_type eq "terminal_left_exon" && $internal_exon_rend == $segment_rend) ) ## a splice boundary in common + ) + { + push (@internal_exons_found, $internal_exon); + } + } + return (@internal_exons_found); +} + +sub _find_terminal_exons_encompassing_segment { + my $self = shift; + my ($lend, $rend, $orient) = @_; + + my @nodes_encompassing_segment; + foreach my $node_obj (@{$self->{_terminal_left_exons}}, @{$self->{_terminal_right_exons}}) { + my ($node_lend, $node_rend) = $node_obj->get_coords(); + my $node_orient = $node_obj->get_orient(); + if ( ($node_orient eq $orient || $orient eq '?') && + ($lend + $FUZZ_DIST >= $node_lend) && + ($rend - $FUZZ_DIST <= $node_rend) ) { + push (@nodes_encompassing_segment, $node_obj); + } + } + + return (@nodes_encompassing_segment); +} + + + + +#### +sub get_assemblies { + my $self = shift; + return (@{$self->{_assemblies}}); +} + +#### +sub get_incoming_alignments { + my $self = shift; + return (@{$self->{_incoming_alignments}}); +} + +#### +sub get_unincorporated_alignments { + my $self = shift; + return (@{$self->{_unincorporated_alignments}}); +} + + +#### +sub _add_alignment_assembly { + my $self = shift; + my (@cdna_alignments) = @_; + + push (@{$self->{_assemblies}}, @cdna_alignments); + return; +} + + + +#### +sub set_incoming_alignments { + my $self = shift; + my @alignments = @_; + @{$self->{_incoming_alignments}} = @alignments; + return; +} + + +#### +sub _decompose_unambiguous_alignment_structures_add_nodes { + my $self = shift; + my ($alignment) = @_; + + my $spliced_orient = $alignment->get_spliced_orientation(); + + my @path_nodes; + + ## Add internal exons + my @segments = $alignment->get_alignment_segments(); + foreach my $segment (@segments) { + #print "seg: " . $segment->toString() . "\n"; + if ($segment->is_internal()) { + my ($exon_lend, $exon_rend) = $segment->get_coords(); + my $node = $self->_add_internal_exon($exon_lend, $exon_rend, $spliced_orient); + push (@path_nodes, $node); + } + } + + ## Add introns + my @intron_coords = $alignment->get_intron_coords(); + + foreach my $intron_coordset (@intron_coords) { + my ($intron_lend, $intron_rend) = @$intron_coordset; #already sorted + my $node = $self->_add_intron($intron_lend, $intron_rend, $spliced_orient); + push (@path_nodes, $node); + } + + ## add a path + $self->_extract_and_add_path_from_node_list(@path_nodes); + + + return; +} + + +#### +sub _extract_and_add_path_from_node_list { + my $self = shift; + my @path_nodes = @_; + + ## sort by genome order: + @path_nodes = sort {$a->{lend} <=> $b->{lend}} @path_nodes; + + my @ordered_nodeID_list; + my @coords; + foreach my $path_node (@path_nodes) { + my $nodeID = $path_node->get_nodeID(); + push (@ordered_nodeID_list, $nodeID); + push (@coords, $path_node->get_coords()); + + } + my $orient = $path_nodes[0]->get_orient(); + @coords = sort {$a<=>$b} @coords; + my $lend = shift @coords; + my $rend = pop @coords; + my $splice_graph_path = Splice_graph_path->new($lend, $rend, $orient, \@ordered_nodeID_list); + + $self->_add_splice_graph_path($splice_graph_path); + + return; +} + +#### +sub _try_terminal_exon_addition { + my $self = shift; + my ($exon_type, $segment, $orient) = @_; + + my ($segment_lend, $segment_rend) = $segment->get_coords(); + + print "Examining terminal exon: $exon_type, $segment_lend, $segment_rend\n" if $SEE; + + ## only consider this a genuine terminal exon if it doesn't appear to be part of an + ## existing internal exon from a more complete alignment + if (my @internal_segments = $self->_find_internal_exons_encompassing_coords_and_share_boundary($exon_type, $segment_lend, $segment_rend, $orient)) { + if ($SEE) { + print "$exon_type, $segment_lend-$segment_rend, $orient, found already represented by internal exons:\n"; + foreach my $internal_segment (@internal_segments) { + print "\t" . $internal_segment->toString() . "\n"; + } + } + return; + } + + my $exon = $self->_find_existing_terminal_exon ($exon_type, $segment_lend, $segment_rend, $orient); + if ($exon) { + print "-found existing terminal exon with shared boundary: " . $exon->toString() . "\n" if $SEE; + my $nodeID = $exon->get_nodeID(); + ## check for boundary adjustment: + my ($exon_lend, $exon_rend) = $exon->get_coords(); + my $nonsplice_coord = undef; + if ($exon_type eq "terminal_left_exon") { + $nonsplice_coord = $segment_lend; + if ($segment_lend < $exon_lend) { + print "-extending left boundary to $segment_lend\n" if $SEE; + $exon->set_coords($segment_lend, $exon_rend); # extend left boundary + } + } + elsif ($exon_type eq "terminal_right_exon") { + $nonsplice_coord = $segment_rend; + if ($segment_rend > $exon_rend) { + print "-extending right boundary to $segment_rend\n" if $SEE; + $exon->set_coords($exon_lend, $segment_rend); + } + } + else { + confess "Error, exon type $exon_type not accounted for. "; # should never get here anyway + } + $exon->increment_evidence_support(); ## account for extra evidence + $self->_add_to_terminal_exon_nonsplice_position_list($nodeID, $nonsplice_coord); + + } + else { + ## add new terminal exon: + $self->_add_graph_node($exon_type, $segment_lend, $segment_rend, $orient); + } + + return; +} + +#### +sub _add_to_terminal_exon_nonsplice_position_list { + my $self = shift; + my ($nodeID, $nonsplice_coord) = @_; + + my $nodeID_to_pos_list_href = $self->{_terminal_exon_nodeID_to_nonsplice_position_list}; + + my $list_aref = $nodeID_to_pos_list_href->{$nodeID}; + unless (ref $list_aref) { + $list_aref = $nodeID_to_pos_list_href->{$nodeID} = []; + } + push (@$list_aref, $nonsplice_coord); + + return; +} + + +#### +sub _find_existing_terminal_exon { + my $self = shift; + my ($exon_type, $segment_lend, $segment_rend, $orient) = @_; + my $exon_list_aref = undef; + if ($exon_type eq "terminal_left_exon") { + $exon_list_aref = $self->{_terminal_left_exons}; + } + elsif ($exon_type eq "terminal_right_exon") { + $exon_list_aref = $self->{_terminal_right_exons}; + } + else { + confess "exon_type $exon_type not accepted for terminal exons"; + } + + foreach my $exon (@$exon_list_aref) { + my ($lend, $rend) = $exon->get_coords(); + if ( + ($exon->get_orient() eq $orient) && + ( + ($exon_type eq "terminal_left_exon" && $rend == $segment_rend) + || + ($exon_type eq "terminal_right_exon" && $lend == $segment_lend) + ) + ) + { + + return ($exon); # found it! + } + } + + return (undef); # didn't find one. +} + +#### +sub _augment_internal_exon_scores_using_terminal_alignment_segment { + my $self = shift; + my ($exon_type, $segment_lend, $segment_rend, $segment_orient) = @_; + + my @relevant_internal_exons = $self->_find_internal_exons_encompassing_coords_and_share_boundary($exon_type, $segment_lend, $segment_rend, $segment_orient); + foreach my $internal_exon (@relevant_internal_exons) { + + $internal_exon->increment_evidence_support(); + } + + return; +} + +#### +sub _merge_overlapping_right_to_left_terminal_exons { + my $self = shift; + + ## looking for this situation: + # <----------- right terminal exon + # ---------> left terminal exon + # that can be merged into: + # <--------------> an internal exon + # + + my %nodeIDs_targeted_for_removal; # if they fully overlap, construct a nice internal exon w/o extensions beyond splice boundary + + foreach my $right_terminal_exon (@{$self->{_terminal_right_exons}}) { + my $right_orient = $right_terminal_exon->get_orient(); + my ($right_lend, $right_rend) = $right_terminal_exon->get_coords(); + my $right_nodeID = $right_terminal_exon->get_nodeID(); + + foreach my $left_terminal_exon (@{$self->{_terminal_left_exons}}) { + my $left_orient = $left_terminal_exon->get_orient(); + my ($left_lend, $left_rend) = $left_terminal_exon->get_coords(); + my $left_nodeID = $left_terminal_exon->get_nodeID(); + + unless ($right_orient eq $left_orient) { next; } # must be transcribed on same strand! + + ## check for overlap: + unless ($left_lend <= $right_rend && $left_rend >= $right_lend) { next;} + + ## make sure right's left splice is before left's right splice (doh! should have better names). + unless ($right_lend < $left_rend) { next; } + + ## check for extensions: + my $left_overhang = $right_lend - $left_lend; + my $right_overhang = $right_rend - $left_rend; + + my $merge_flag = 0; + if ($left_overhang <= $FUZZ_DIST && $right_overhang <= $FUZZ_DIST) { + ## nice merge, as in illustration + $merge_flag = 1; + ## also, target these for deletion now. + $nodeIDs_targeted_for_removal{$right_nodeID} = 1; + $nodeIDs_targeted_for_removal{$left_nodeID} = 1; + } + else { + ## must check position lists to see if transcripts are contained that have + ## nicely overlapping boundaries + if ($self->_right_left_terminal_exons_overlap_via_position_lists($right_nodeID, $left_nodeID, $right_lend, $left_rend)) { + $merge_flag = 1; + if ($left_overhang <= $FUZZ_DIST) { ## check for insufficient overhang + $nodeIDs_targeted_for_removal{$left_nodeID} = 1; + } + if ($right_overhang <= $FUZZ_DIST) { + $nodeIDs_targeted_for_removal{$right_nodeID} = 1; + } + } + } + if ($merge_flag) { + $self->_merge_terminal_exons($right_terminal_exon, $left_terminal_exon); + } + } + } + + ## process deletions: + if (%nodeIDs_targeted_for_removal) { + $self->_purge_nodes(%nodeIDs_targeted_for_removal); + } + + return; +} + +#### +sub _merge_terminal_exons { + my $self = shift; + my ($right_terminal_exon, $left_terminal_exon) = @_; + + my ($right_lend, $right_rend) = $right_terminal_exon->get_coords(); + my ($left_lend, $left_rend) = $left_terminal_exon->get_coords(); + + my ($exon_lend, $exon_rend) = ($right_lend, $left_rend); ## splice junctions for new internal exon + my $orient = $right_terminal_exon->get_orient(); + if ($orient ne $left_terminal_exon->get_orient()) { + confess "Error, trying to merge two terminal exons with opposite transcriptional orientations!"; + } + + if ($self->_graph_node_exists("internal_exon", $exon_lend, $exon_rend, $orient)) { + confess "Error, trying to merge two terminal exons into an internal exon that already exists!"; + } + + my $node = $self->_add_graph_node("internal_exon", $exon_lend, $exon_rend, $orient); + ## add to it the evidence from the other nodes being merged: + my $evidence_support_to_add = $left_terminal_exon->get_num_evidence_support() + $right_terminal_exon->get_num_evidence_support(); + $evidence_support_to_add -= 1; # node already has value of one. + if ($evidence_support_to_add) { + $node->increment_evidence_support($evidence_support_to_add); + } + return; +} + + +#### +sub _add_splice_graph_path { + my $self = shift; + my ($splice_graph_path) = @_; + + ## add it as long as: + # -it's not a subpath of an already existing path + # -if an existing path is a subpath of this, remove it and replace it with this path + + my @current_valid_splice_paths = $self->_get_valid_splice_paths(); + + ## check to see if splice_graph_path is already represented in the current path list: + foreach my $current_valid_splice_path (@current_valid_splice_paths) { + if ($splice_graph_path->is_subpath_of($current_valid_splice_path)) { + return; # nothing to do; path is already included as a subset of the current path set + } + } + + # if got this far, our new path is not already fully represented. + # add it, and any other existing splice paths that are not a subpath of it. + my @new_valid_splice_paths = ($splice_graph_path); + foreach my $current_valid_splice_path (@current_valid_splice_paths) { + if (! $current_valid_splice_path->is_subpath_of($splice_graph_path)) { + push (@new_valid_splice_paths, $current_valid_splice_path); + } + } + + $self->_set_valid_splice_paths(@new_valid_splice_paths); + + return; +} + +#### +sub _get_valid_splice_paths { + my $self = shift; + return (@{$self->{_valid_splice_paths}}); +} + +#### +sub _set_valid_splice_paths { + my $self = shift; + my @valid_splice_paths = @_; + ## completely stomps the existing contents!!! + + @{$self->{_valid_splice_paths}} = @valid_splice_paths; + return; +} + +sub _graph_node_exists { + my $self = shift; + my ($type, $lend, $rend, $orient) = @_; + my $hashkey = $self->_get_hash_key($type, $lend, $rend, $orient); + return (exists $self->{_graph_node_hashkey_lookup}->{$hashkey}); +} + +sub _get_graph_node_via_coords_n_type { + my $self = shift; + my ($type, $lend, $rend, $orient) = @_; + my $hashkey = $self->_get_hash_key($type, $lend, $rend, $orient); + my $node = $self->{_graph_node_hashkey_lookup}->{$hashkey}; + unless ($node) { + confess "Error, no graph node retrieved based on data ($type, $lend, $rend, $orient)"; + } + return ($node); +} + + + +#### +sub _add_graph_node { + my $self = shift; + my ($type, $lend, $rend, $orient) = @_; + my $hash_key = $self->_get_hash_key($type, $lend, $rend, $orient); + + ## only add the node if it doesn't already exist! + my $graph_node_hashkey_lookup_href = $self->{_graph_node_hashkey_lookup}; + if (my $existing_node = $graph_node_hashkey_lookup_href->{$hash_key}) { + ## increment evidence for existing node: + $existing_node->increment_evidence_support(); + return ($existing_node); + } + else { + # add it + my $node = Splice_graph_node->new($type, $lend, $rend, $orient); + push (@{$self->{_graph_nodes}}, $node); # add to complete node list + ## helpers for node access: + $graph_node_hashkey_lookup_href->{$hash_key} = $node; # store in lookup table + my $nodeID = $node->get_nodeID(); + $self->{_graph_node_via_nodeID}->{$nodeID} = $node; + + ## splay based on type: + my $type = $node->get_type(); + + if ($type eq "internal_exon") { + push (@{$self->{_internal_exons}}, $node); + } + elsif ($type eq "intron") { + push (@{$self->{_introns}}, $node); + } + elsif ($type eq "terminal_left_exon") { + push (@{$self->{_terminal_left_exons}}, $node); + } + elsif ($type eq "terminal_right_exon") { + push (@{$self->{_terminal_right_exons}}, $node); + } + elsif ($type eq "singleton_exon") { + push (@{$self->{_singleton_exons}}, $node); + } + else { + confess "Error, do not recognize type: $type for node"; + } + return ($node); + } + +} + +#### +sub _add_intron { + my $self = shift; + my ($intron_lend, $intron_rend, $spliced_orient) = @_; + + my $node = $self->_add_graph_node("intron", $intron_lend, $intron_rend, $spliced_orient); + return ($node); +} + +#### +sub _add_internal_exon { + my $self = shift; + my ($exon_lend, $exon_rend, $spliced_orient) = @_; + my $node = $self->_add_graph_node("internal_exon", $exon_lend, $exon_rend, $spliced_orient); + return ($node); +} + + +#### +sub get_graph_nodes { + my $self = shift; + return (sort {$a->{lend}<=>$b->{lend}} @{$self->{_graph_nodes}}); +} + + + + +#### +sub get_graph_node_via_nodeID { + my $self = shift; + my $nodeID = shift; + my $node_obj = $self->{_graph_node_via_nodeID}->{$nodeID} or confess "Error, no node found based on nodeID: $nodeID\n" . $self->toString(); + return ($node_obj); +} + +#### +sub toString { + my $self = shift; + my @graph_nodes = $self->get_graph_nodes(); + my $num_graph_nodes = scalar (@graph_nodes); + my $text = "Splice_graph_assembler instance with $num_graph_nodes graph nodes:\n"; + foreach my $graph_node (@graph_nodes) { + $text .= $graph_node->toString() . "\n"; + } + + $text .= "\tvalid paths thru nodes:\n"; + foreach my $splice_path ($self->_get_valid_splice_paths()) { + $text .= "\t" . $splice_path->toString() . "\n"; + } + + $text .= "\tassembled splice paths:\n"; + foreach my $splice_path (@{$self->{_assembled_splice_paths}}) { + $text .= "\t" . $splice_path->toString() . "\n"; + } + + $text .= "\tFinal assemblies with termini\n"; + foreach my $assembly ($self->get_assemblies()) { + my @node_list = @{$assembly->{__Splice_graph_assembler_nodeID_list}}; + $text .= "\t" . join (",", @node_list) . "\n"; + } + + $text .= $self->toAlignIllustration(60); + + return ($text); + +} + +=item toAlignIllustration() + +=over 4 + +B illustrates the individual cDNAs to be assembled along with the final products. + +B $max_line_chars(optional) + +$max_line_chars is an integer representing the maximum number of characters in a single line of output to the terminal. The default is 100. + +B $alignment_illustration_text + +$alignment_illustration_text is a string containing a paragraph of text which illustrates the alignments and assemblies. An example is below: + + ---> <--> <-----> <---> <---------------- (+)gi|1199466 + + ---> <--> <-----> <---> <------------ (+)gi|1209702 + +----> <--> <---- (+)AV827070 + +----> <--> <--- (+)AV828861 + +----> <--> <--- (+)AV830936 + + ---> <--> <- (+)H36350 + +ASSEMBLIES: (1) + +----> <--> <-----> <---> <---------------- (+) gi|1199466, gi|1209702, AV827070, AV828861, AV830936, H36350 + + + + +=back + +=cut + + ; + +sub toAlignIllustration () { + my $self = shift; + my $max_line_chars = shift; + $max_line_chars = ($max_line_chars) ? $max_line_chars : 100; #if not specified, 100 chars / line is default. + + ## Get minimum coord for relative positioning. + my @coords; + my @alignments = @{$self->{_incoming_alignments}}; + foreach my $alignment (@alignments) { + my @c = $alignment->get_coords(); + push (@coords, @c); + } + @coords = sort {$a<=>$b} @coords; + print "coords: @coords\n" if $::SEE; + my $min_coord = shift @coords; + my $max_coord = pop @coords; + my $rel_max = $max_coord - $min_coord; + my $alignment_text = ""; + ## print each alignment followed by assemblies: + my $num_alignments = $#alignments + 1; + $alignment_text .= "Individual Alignments: ($num_alignments)\n"; + my $i = 0; + foreach my $alignment (@alignments) { + $alignment_text .= (sprintf ("%3d ", $i)) . $alignment->toAlignIllustration($min_coord, $rel_max, $max_line_chars) . "\n"; + $i++; + } + + my @assemblies = @{$self->{_assemblies}}; + my $num_assemblies = $#assemblies + 1; + $alignment_text .= "\n\nASSEMBLIES: ($num_assemblies)\n"; + foreach my $assembly (@assemblies) { + $alignment_text .= " " . $assembly->toAlignIllustration($min_coord, $rel_max, $max_line_chars) . "\n"; + } + + if (my @unincorporated_alignments = $self->get_unincorporated_alignments()) { + my $num_unincorporated = scalar @unincorporated_alignments; + $alignment_text .= "\n\nUNINCORPORATED_ALIGNMENTS($num_unincorporated)\n"; + foreach my $alignment (@unincorporated_alignments) { + $alignment_text .= " " . $alignment->toAlignIllustration($min_coord, $rel_max, $max_line_chars) . "\n"; + } + } + + return ($alignment_text); +} + +#### +sub _get_hash_key { + my $self = shift; + my ($type, $lend, $rend, $orient) = @_; + return ("$type,$lend,$rend,$orient"); +} + +#### +sub _purge_nodes { + my $self = shift; + my (%nodeIDs) = @_; + + ## perform deletions based on hashkeys + foreach my $nodeID (keys %nodeIDs) { + print "PURGING node: $nodeID\n" if $SEE; + my $node = $self->{_graph_node_via_nodeID}->{$nodeID}; + my $type = $node->get_type(); + my ($lend, $rend) = $node->get_coords(); + my $orient = $node->get_orient(); + + my $hashkey = $self->_get_hash_key($type, $lend, $rend, $orient); + delete $self->{_graph_node_hashkey_lookup}->{$hashkey}; + delete $self->{_graph_node_via_nodeID}->{$nodeID}; + } + + ## now, do array replacements: + foreach my $node_list_aref ( $self->{_graph_nodes}, + ## only deleting the terminal exons, even though this could be more generic + $self->{_terminal_left_exons}, + $self->{_terminal_right_exons}, + ) { + my @replacments; + my $need_replacement_flag = 0; + foreach my $node (@$node_list_aref) { + my $nodeID = $node->get_nodeID(); + if ($nodeIDs{$nodeID}) { + ## must delete! + $need_replacement_flag = 1; + } + else { + push (@replacments, $node); + } + } + if ($need_replacement_flag) { + @$node_list_aref = @replacments; ## Doing Replacment + } + } + + return; +} + +#### +sub _right_left_terminal_exons_overlap_via_position_lists { + my $self = shift; + my ($right_nodeID, $left_nodeID, $left_splice_coord, $right_splice_coord) = @_; + + ## looking for the following: + # + # ## termini of transcripts in right terminal + # <---------------------X---------------------- ## right terminal exon + # -------------------X---------------------------> ## left terminal exon + # ## termini of transcripts in left terminal + # + # + # The X termini show the endpoints of other transcripts incorporated that would define a proper merging situation. + + # look for boundary pair such that both are included within the splice junctions + # and right boundary >= left boundary + + my $right_pos_list_aref = $self->{_terminal_exon_nodeID_to_nonsplice_position_list}->{$right_nodeID}; + my $left_pos_list_aref = $self->{_terminal_exon_nodeID_to_nonsplice_position_list}->{$left_nodeID}; + + foreach my $right_pos (@$right_pos_list_aref) { + unless ($right_pos > $left_splice_coord && $right_pos < $right_splice_coord) { next; } + + foreach my $left_pos (@$left_pos_list_aref) { + unless ($left_pos > $left_splice_coord && $left_pos < $right_splice_coord) { next; } + + if ($right_pos >= $left_pos) { + return (1); # found suitable case + } + } + } + + return (0); # no such example found. +} + +#### +sub _analyze_intronless_alignments { + my $self = shift; + my @intronless_segments = @_; ## actually alignment objects. + + + print "method: _analyze_intronless_alignments()\n" if $SEE; + + my %applied_segment_indices; + + { + ## hack in a hidden attribute that uniquely identifies each of these segments. + ## perl allows this, but I'm not so happy with doing it. for now, it'll be fine. + ## This __intronless_segment_index will be used to identify those segments that are + ## applied to existing structures (internal/terminal exons), and those left over + ## and still need to be accounted for. + + my $id = 0; + foreach my $intronless_segment (@intronless_segments) { + $id++; + $intronless_segment->{__intronless_segment_ID} = $id; + } + } + + ## increment evidence for internal exons containing intronless segment + ## and extend terminal exons overlapping intronless segments + + ## first, examine the internal_segments: + foreach my $internal_exon (@{$self->{_internal_exons}}) { + my $internal_exon_orient = $internal_exon->get_orient(); + my ($internal_exon_lend, $internal_exon_rend) = $internal_exon->get_coords(); + + + + foreach my $intronless_segment (@intronless_segments) { + my $intronless_segment_orient = $intronless_segment->get_spliced_orientation(); + my ($intronless_lend, $intronless_rend) = $intronless_segment->get_coords(); + + if ($intronless_segment_orient eq '?' || $intronless_segment_orient eq $internal_exon_orient) { + + ## look for encapsulation + if ( ($intronless_lend + $FUZZ_DIST) >= $internal_exon_lend && + ($intronless_rend - $FUZZ_DIST) <= $internal_exon_rend) { + + $internal_exon->increment_evidence_support(); + $applied_segment_indices{ $intronless_segment->{__intronless_segment_ID} } = 1; # incorporated already + + print "-intronless segment $intronless_lend, $intronless_rend, $intronless_segment_orient found incorporated in internal exon: " . $internal_exon->toString() . "\n" if $SEE; + } + } + } + } + + ## examine left terminal exons + ## first sort so we examine the right boundaries in order from right to left to faciliate extensions + @intronless_segments = reverse sort {$a->{rend}<=>$b->{rend}} @intronless_segments; + + foreach my $left_terminal_exon (@{$self->{_terminal_left_exons}}) { + my $left_terminal_exon_orient = $left_terminal_exon->get_orient(); + + # -----------------> # left terminal exon + + foreach my $intronless_segment (@intronless_segments) { + + my ($left_terminal_lend, $left_terminal_rend) = $left_terminal_exon->get_coords(); # do it here because it might change below + my $intronless_segment_orient = $intronless_segment->get_spliced_orientation(); + my ($intronless_lend, $intronless_rend) = $intronless_segment->get_coords(); + + unless ($intronless_segment_orient eq '?' || $intronless_segment_orient eq $left_terminal_exon_orient) { next; } + + ## must overlap + unless ($left_terminal_lend <= $intronless_rend && $left_terminal_rend >= $intronless_lend) { next; } + + unless ($intronless_rend - $FUZZ_DIST <= $left_terminal_rend) { next; } + ## must be : + # -----------------------> # left terminal exon + # ---------------- # intronless segment, overlaps but not passed the splice boundary + + # extend lend if intronless segment passes it + if ($intronless_lend < $left_terminal_lend) { + # update coords: + $left_terminal_exon->set_coords($intronless_lend, $left_terminal_rend); + } + + ## if got this far, evidence is incorporated. + $left_terminal_exon->increment_evidence_support(); + $applied_segment_indices{ $intronless_segment->{__intronless_segment_ID} } = 1; # incorporated + print "-intronless segment $intronless_lend, $intronless_rend, $intronless_segment_orient incorporated in left terminal exon: " . $left_terminal_exon->toString() . "\n" if $SEE; + } + } + + ## examine right terminal exons + ## first sort so we examine the left boundaries in order from left to right to faciliate extensions + @intronless_segments = sort {$a->{lend}<=>$b->{lend}} @intronless_segments; + + foreach my $right_terminal_exon (@{$self->{_terminal_right_exons}}) { + my $right_terminal_exon_orient = $right_terminal_exon->get_orient(); + + # <----------------- # right terminal exon + + foreach my $intronless_segment (@intronless_segments) { + + my ($right_terminal_lend, $right_terminal_rend) = $right_terminal_exon->get_coords(); # do it here because it might change below + my $intronless_segment_orient = $intronless_segment->get_spliced_orientation(); + my ($intronless_lend, $intronless_rend) = $intronless_segment->get_coords(); + + unless ($intronless_segment_orient eq '?' || $intronless_segment_orient eq $right_terminal_exon_orient) { next; } + + ## must overlap + unless ($right_terminal_lend <= $intronless_rend && $right_terminal_rend >= $intronless_lend) { next; } + + ## must be : + # <---------------- # right terminal exon + # ---------------- # intronless segment, overlaps but not passed the splice boundary + + unless ($intronless_lend + $FUZZ_DIST >= $right_terminal_lend) { next; } + + # extend lend if intronless segment passes it + if ($intronless_rend > $right_terminal_rend) { + # update coords: + $right_terminal_exon->set_coords($right_terminal_lend, $intronless_rend); + } + + ## if got this far, evidence is incorporated. + $right_terminal_exon->increment_evidence_support(); + $applied_segment_indices{ $intronless_segment->{__intronless_segment_ID} } = 1; # incorporated + print "-intronless segment $intronless_lend, $intronless_rend, $intronless_segment_orient incorporated in right terminal exon: " . $right_terminal_exon->toString() . "\n" if $SEE; + } + } + + ## check to see if there are any unincorporated single segments, and if so, instantiate maximal single segments that contain them. + my $have_leftover_single_segments_flag = 0; + foreach my $intronless_segment (@intronless_segments) { + unless ($applied_segment_indices{ $intronless_segment->{__intronless_segment_ID} }) { + $have_leftover_single_segments_flag = 1; + last; + } + } + unless ($have_leftover_single_segments_flag) { + ## all done; they've all been accounted for. + print "-All intronless segments are accounted for.\n" if $SEE; + return; + } + + ## if got here, we have some single segments that haven't been accounted for. + print "-some intronless segments are as of yet unincorporated. Need to instantiate singleton exons\n" if $SEE; + $self->_instantiate_maximal_single_exons(\@intronless_segments, \%applied_segment_indices); + + return; +} + +#### +sub _instantiate_maximal_single_exons { + my $self = shift; + my ($intronless_alignments_aref, $applied_segment_indices_href) = @_; + + print "\nmethod: _instantiate_maximal_single_exons\n" if $SEE; + + ## get all coords for segments: + my %segment_id_to_coords; + my %segment_id_to_alignment; + my $plus_strand_overlap_assembler = new Overlap_piler(); + my $minus_strand_overlap_assembler = new Overlap_piler(); + my $got_plus_strand_flag = 0; + my $got_minus_strand_flag = 0; + + foreach my $alignment (@$intronless_alignments_aref) { + print "Got intronless alignment: " . $alignment->toToken() . "\n" if $SEE; + my ($align_lend, $align_rend) = $alignment->get_coords(); + my $align_id = $alignment->{__intronless_segment_ID}; + $segment_id_to_coords{$align_id} = [$align_lend, $align_rend]; + $segment_id_to_alignment{$align_id} = $alignment; + + my $spliced_orient = $alignment->get_spliced_orientation(); + if ($spliced_orient eq '+' || $spliced_orient eq '?') { + $plus_strand_overlap_assembler->add_coordSet($align_id, $align_lend, $align_rend); + print "Adding $align_lend-$align_rend [s$spliced_orient] to plus strand assembler\n" if $SEE; + if ($spliced_orient eq '+') { + $got_plus_strand_flag = 1; + } + } + if ($spliced_orient eq '-' || $spliced_orient eq '?') { + $minus_strand_overlap_assembler->add_coordSet($align_id, $align_lend, $align_rend); + print "Adding $align_lend-$align_rend [s$spliced_orient] to minus strand assembler\n" if $SEE; + if ($spliced_orient eq '-') { + $got_minus_strand_flag = 1; + } + } + + } + + ## unless we have some known spliced orientation, then just assemble everything using the plus strand assembler; either could be used w/ no difference. + unless ($got_plus_strand_flag || $got_minus_strand_flag) { + $got_plus_strand_flag = 1; + } + + + my @singleton_clusters; + if ($got_plus_strand_flag) { + print "\nBuilding Plus strand clusters of singletons:\n" if $SEE; + my @plus_strand_clusters = $plus_strand_overlap_assembler->build_clusters(); + if ($SEE) { + print "\nSingleton assemblies, PLUS strand assembler:\n"; + @plus_strand_clusters = sort {$a->[0]<=>$b->[0]} @plus_strand_clusters; + foreach my $cluster (@plus_strand_clusters) { + print "clustered IDs [+]: " . join ("-", @$cluster) . "\n"; + } + } + push (@singleton_clusters, @plus_strand_clusters); + } + if ($got_minus_strand_flag) { + print "\nBuilding Minus strand clusters of singletons:\n" if $SEE; + my @minus_strand_clusters = $minus_strand_overlap_assembler->build_clusters(); + if ($SEE) { + print "\nSingleton assemblies, MINUS strand assembler:\n"; + @minus_strand_clusters = sort {$a->[0]<=>$b->[0]} @minus_strand_clusters; + foreach my $cluster (@minus_strand_clusters) { + print "clustered IDs [-]: " . join ("-", @$cluster) . "\n"; + } + } + push (@singleton_clusters, @minus_strand_clusters); + } + + ## convert to singleton assemblies: + my @singleton_assemblies; + foreach my $singleton_cluster (@singleton_clusters) { + my @ids = @$singleton_cluster; + my ($min_lend, $max_rend); + my %aligned_orient_counts; + my %spliced_orient_counts; + foreach my $id (@ids) { + my ($lend, $rend) = @{$segment_id_to_coords{$id}}; + if (! defined ($min_lend)) { + ($min_lend, $max_rend) = ($lend, $rend); + } + else { + if ($lend < $min_lend) { $min_lend = $lend; } + if ($rend > $max_rend) { $max_rend = $rend; } + } + my $alignment = $segment_id_to_alignment{$id}; + my $spliced_orient = $alignment->get_spliced_orientation(); + my $aligned_orient = $alignment->get_aligned_orientation(); + if ($spliced_orient ne '?') { + $spliced_orient_counts{$spliced_orient}++; + } + $aligned_orient_counts{$aligned_orient}++; + } + + ## get orientation value with most support: + my @spliced_orients = reverse sort {$spliced_orient_counts{$a}<=>$spliced_orient_counts{$b}} keys %spliced_orient_counts; + if (scalar @spliced_orients > 1) { + # should only be one spliced orient, or no spliced orient if everything was '?' + confess "Error, captured a singleton assembly with multiple spliced orientations."; + } + my $assembly_spliced_orient = shift @spliced_orients; + + unless ($assembly_spliced_orient) { + ## no evidence for assembly spliced orientation. + ## take the aligned orientation that has the greatest support. + my @aligned_orients = reverse sort {$aligned_orient_counts{$a}<=>$aligned_orient_counts{$b}} keys %aligned_orient_counts; + $assembly_spliced_orient = shift @aligned_orients; + } + + unless ($assembly_spliced_orient =~ /^[\+\-]$/) { + confess "Error, couldn't decide upon assembly orientation for singleton assembly"; + } + + push (@singleton_assemblies, { lend => $min_lend, + rend => $max_rend, + ids => [@ids], + length => $max_rend - $min_lend + 1, + num_alignments_included => scalar @ids, + orient => $assembly_spliced_orient, + } + ); + } + + ## sort so that we maximize for number of alignments included and alignment span + @singleton_assemblies = reverse sort {$a->{num_alignments_included}<=>$b->{num_alignments_included} + || + $a->{length}<=>$b->{length}} @singleton_assemblies; + + my %ids_accounted_for; + + SINGLETON_ASSEMBLY_ANALYSIS: + foreach my $singleton_assembly (@singleton_assemblies) { + my ($lend, $rend, $orient, $ids_aref) = ($singleton_assembly->{lend}, + $singleton_assembly->{rend}, + $singleton_assembly->{orient}, + $singleton_assembly->{ids}); + + + ## make sure singleton contains an alignment that wasn't already applied elsewhere: + my $has_alignment_not_applied_elsewhere_flag = 0; + foreach my $id (@$ids_aref) { + unless ($applied_segment_indices_href->{$id}) { + $has_alignment_not_applied_elsewhere_flag = 1; + last; + } + } + unless ($has_alignment_not_applied_elsewhere_flag) { + ## no reason to bother pursuing it. Everythings accounted for in other already instantiated exons. + next SINGLETON_ASSEMBLY_ANALYSIS; + } + + + my $found_id_not_accounted_for_flag = 0; ## track those IDs that are built into other already instantiated singleton assemblies + foreach my $id (@$ids_aref) { + unless ($ids_accounted_for{$id}) { + $found_id_not_accounted_for_flag = 1; + last; + } + } + if ($found_id_not_accounted_for_flag) { + ## add a singleton assembly: + $self->_add_graph_node("singleton_exon", $lend, $rend, $orient); + foreach my $id (@$ids_aref) { + $ids_accounted_for{$id} = 1; + } + } + } + + + ## Ensure that all are accounted for now: + foreach my $intronless_alignment (@$intronless_alignments_aref) { + my $id = $intronless_alignment->{__intronless_segment_ID}; + if (! $applied_segment_indices_href) { + ## if not applied before entering this method, then should be applied now + if (! $ids_accounted_for{$id}) { + my $unaccounted_for_intronless_alignment = $segment_id_to_alignment{$id}; + confess "Error, intronless alignment " . $unaccounted_for_intronless_alignment->toToken() . "\n" + . "is not accounted for after instantiating maximal single exons.\n"; + } + } + } + print "-all intronless alignments should now be accounted for.\n" if $SEE; + + return; +} + + +#### +sub _chain_compatible_splice_paths { + my $self = shift; + + ## wrap splice path objs into structs for scoring and dynamic programming to build chains: + my @path_structs; + foreach my $splice_path ($self->_get_valid_splice_paths()) { + my $pathID = $splice_path->get_pathID(); + my @nodeIDs = $splice_path->get_ordered_nodeIDs(); + my $evidence_support_count = $self->_get_evidence_support_from_nodeIDs(@nodeIDs); + my $orient = $splice_path->get_orient(); + my ($lend, $rend) = $splice_path->get_coords(); + + my $struct = { splice_path_obj => $splice_path, + path_score => $evidence_support_count, + struct_score => $evidence_support_count, + nodeIDs => [@nodeIDs], + lend => $lend, + rend => $rend, + orient => $orient, + pathID => $pathID, + prev => undef, # previous link in a chain of compatible structs + }; + push (@path_structs, $struct); + } + + @path_structs = sort {$a->{lend}<=>$b->{lend}} @path_structs; + + ## score them: + for (my $i = 0; $i < $#path_structs; $i++) { + + my $i_path_struct = $path_structs[$i]; + my ($i_lend, $i_rend, $i_orient, $i_splice_path, $i_nodeIDs_aref) = ($i_path_struct->{lend}, + $i_path_struct->{rend}, + $i_path_struct->{orient}, + $i_path_struct->{splice_path_obj}, + $i_path_struct->{nodeIDs}); + + + for (my $j = $i+1; $j <= $#path_structs; $j++) { + + ## No self comparisons: + if ($i == $j) { next; } + + ## struct j comes after struct i + my $j_path_struct = $path_structs[$j]; + my ($j_lend, $j_rend, $j_orient, $j_splice_path, $j_nodeIDs_aref) = ($j_path_struct->{lend}, + $j_path_struct->{rend}, + $j_path_struct->{orient}, + $j_path_struct->{splice_path_obj}, + $j_path_struct->{nodeIDs}); + + ## must have same orient: + unless ($i_orient eq $j_orient) { next; } + + ## check for overlap: + unless ($i_lend <= $j_rend && $i_rend >= $j_lend) { next; } + + ## must be compatible: + unless ($i_splice_path->is_compatible($j_splice_path)) { next; } + + ## double check that neither is a subset of the other: + if ($i_splice_path->is_subpath_of($j_splice_path) || $j_splice_path->is_subpath_of($i_splice_path)) { + confess "Error, trying to chain two splice paths where one is a subpath of the other:\n" . $self->toString() + . "\nOffending paths:\n" + . $i_splice_path->toString() . "\n" + . $j_splice_path->toString() . "\n"; + } + + ## calculate path score ending at struct j + ## don't count the same nodes twice, though: + my @nodes_found_in_both_paths = $self->_intersection_of_lists($i_nodeIDs_aref, $j_nodeIDs_aref); + my $score_decrement = $self->_get_evidence_support_from_nodeIDs(@nodes_found_in_both_paths); + my $candidate_path_score_j = $i_path_struct->{path_score} + $j_path_struct->{struct_score} - $score_decrement; + if ($candidate_path_score_j > $j_path_struct->{path_score}) { + ## found best path: + $j_path_struct->{path_score} = $candidate_path_score_j; + $j_path_struct->{prev} = $i_path_struct->{pathID}; + } + } + } + + ## get the highest scoring paths of compatible subpaths (structs) + my %pathIDs_left; + my %pathID_to_struct; + foreach my $struct (@path_structs) { + my $pathID = $struct->{pathID}; + $pathIDs_left{$pathID} = 1; + $pathID_to_struct{$pathID} = $struct; + } + + my @final_assembled_path_node_lists; + while (%pathIDs_left) { + ## find the highest scoring chain of paths: + my $highest_path_score = 0; + my $highest_scoring_pathID = undef; + + foreach my $pathID (keys %pathIDs_left) { + my $struct = $pathID_to_struct{$pathID}; + my $path_score = $struct->{path_score}; + if ($path_score > $highest_path_score) { + $highest_path_score = $path_score; + $highest_scoring_pathID = $pathID; + } + } + + my %nodes_along_chain; + ## walk along chain links and collect the nodes: + my $pathID = $highest_scoring_pathID; + while (defined $pathID) { + delete $pathIDs_left{$pathID}; # remove it so we know it's accounted for as a traversed chain link. + my $struct = $pathID_to_struct{$pathID}; + my $nodes_aref = $struct->{nodeIDs}; + foreach my $node (@$nodes_aref) { + $nodes_along_chain{$node} = 1; + } + $pathID = $struct->{prev}; + } + + my @nodes_along_chain_list = keys %nodes_along_chain; + push (@final_assembled_path_node_lists, [@nodes_along_chain_list]); + } + + ## Create new path objects for these chains of subpaths: + foreach my $node_list (@final_assembled_path_node_lists) { + my @nodes = @$node_list; + ## sort them according to position: + @nodes = sort { $self->get_graph_node_via_nodeID($a)->{lend} + <=> + $self->get_graph_node_via_nodeID($b)->{lend}} @nodes; + my $path_lend = $self->get_graph_node_via_nodeID($nodes[0])->{lend}; + my $path_rend = $self->get_graph_node_via_nodeID($nodes[$#nodes])->{rend}; + my $orient = $self->get_graph_node_via_nodeID($nodes[0])->get_orient(); + + my $splice_path = Splice_graph_path->new($path_lend, $path_rend, $orient, [@nodes]); + push (@{$self->{_assembled_splice_paths}}, $splice_path); + } + return; +} + +#### +sub _intersection_of_lists { + my $self = shift; + my ($list_A_aref, $list_B_aref) = @_; + + my %eles_in_A; + foreach my $ele (@$list_A_aref) { + $eles_in_A{$ele} = 1; + } + my @intersection; + foreach my $ele (@$list_B_aref) { + if ($eles_in_A{$ele}) { + push (@intersection, $ele); + } + } + + return (@intersection); +} + + +#### +sub _get_evidence_support_from_nodeIDs { + my $self = shift; + my @nodeIDs = @_; + my $evidence_sum = 0; + foreach my $nodeID (@nodeIDs) { + my $node = $self->get_graph_node_via_nodeID($nodeID); + my $ev_support = $node->get_num_evidence_support(); + $evidence_sum += $ev_support; + } + return ($evidence_sum); +} + +#### +sub _build_splice_graph { + my $self = shift; + + print "## building splice graph:\n" if $SEE; + + ## do n^2 comparison of nodes: + ## number of nodes is relatively small, so this isn't too much of a bottleneck + my @nodes = $self->get_graph_nodes(); # already sorted by lend + + ## init base scores for nodes + foreach my $node (@nodes) { + my $ev_support = $node->get_num_evidence_support(); + # init base score + $node->{forward_base_score} = $ev_support; + $node->{reverse_base_score} = $ev_support; + # init path score + $node->{forward_path_score} = $ev_support; + $node->{reverse_path_score} = $ev_support; + } + + ## build the splice graph + ## at the same time, do the forward dynamic programming calculation so we can backtrack to the best scoring path from any node + ## nodeB must come after nodeA, be of same orientation, adjacent, and an acceptable linkage (intron to exon) + ## perform calculation from left to right with backtracking from right to left. + for (my $i = 1; $i <= $#nodes; $i++) { + my $nodeB = $nodes[$i]; + my ($nodeB_lend, $nodeB_rend) = $nodeB->get_coords(); + my $nodeB_orient = $nodeB->get_orient(); + my $nodeB_type = $nodeB->get_type(); + my $nodeB_ID = $nodeB->get_nodeID(); + + for (my $j = $i - 1; $j >= 0; $j--) { + my $nodeA = $nodes[$j]; + my ($nodeA_lend, $nodeA_rend) = $nodeA->get_coords(); + my $nodeA_orient = $nodeA->get_orient(); + my $nodeA_type = $nodeA->get_type(); + my $nodeA_ID = $nodeA->get_nodeID(); + + print "-comparing " . $nodeA->toString() . " <=> " . $nodeB->toString() if $SEE; + + unless ($nodeA_orient eq $nodeB_orient) { + print "-opposite orients, cannot link.\n" if $SEE; + next; + } + + unless ($nodeB_lend == $nodeA_rend + 1) { ## next base coordinate for next feature + print "-not adjacent, cannot link.\n" if $SEE; + next; + } + + ## ensure proper connection: + unless ($ACCEPTABLE_CONNECTIONS{$nodeA_type}->{$nodeB_type}) { + print "-unacceptable linkage types, cannot link ($nodeA_type,$nodeB_type).\n" if $SEE; + next; + } + + print "* linking nodes.\n" if $SEE; + + ## if got this far, connections are perfect! + $nodeA->connect_this_next_nodeID($nodeB_ID); + $nodeB->connect_this_prev_nodeID($nodeA_ID); + + ## check scores for best scoring path solution + my $path_score = $nodeA->{forward_path_score} + $nodeB->{forward_base_score}; + if ($path_score > $nodeB->{forward_path_score}) { + ## link the nodes as current best path linkage + $nodeB->{forward_path_score} = $path_score; + $nodeB->{forward_backtrack_nodeID} = $nodeA_ID; + } + } + } + + ## Do the reverse dynamic programming calculation so we can find the best scoring path in a left to right traversal from any node + ## calculation from right to left with backtracking from left to right + for (my $i = ($#nodes - 1); $i >= 0; $i--) { + my $nodeA = $nodes[$i]; + my $nodeA_ID = $nodeA->get_nodeID(); + + for (my $j = $i+1; $j <= $#nodes; $j++) { + my $nodeB = $nodes[$j]; + my $nodeB_ID = $nodeB->get_nodeID(); + print "reverse comparison " . $nodeA->toString() . " <=> " . $nodeB->toString() . "\t" if $SEE; + ## already stored the prev/next links, so rely on that to know if an acceptable linkage exists: + if ($nodeA->has_next_nodeID($nodeB_ID)) { + print "connectable.\n" if $SEE; + ## check scores for DP + my $base_score = $nodeA->{reverse_base_score}; + my $path_score = $nodeB->{reverse_path_score} + $base_score; + if ($path_score > $nodeA->{reverse_path_score}) { + ## make it the best path and link the node + $nodeA->{reverse_path_score} = $path_score; + $nodeA->{reverse_backtrack_nodeID} = $nodeB_ID; + } + } + else { + print "non-connectable nodes.\n" if $SEE; + } + } + } + + ## defensive programming: + foreach my $node (@nodes) { + + ## ignore singletons + if ($node->get_type() eq "singleton_exon") { next; } + + ## ensure that only terminal left exons have forward_backtrack_nodeID's as undef + if ( (! defined $node->{forward_backtrack_nodeID}) && $node->get_type() ne "terminal_left_exon") { + print $self->toString(); + print Dumper ($node); + confess "Error, node has undefined forward backtrack ID but is not a terinal left exon: " . $node->toString() . "\n"; + } + ## ensure that only terminal right exons have the reverse_backtrack_nodeID's as undef + if ( (! defined $node->{reverse_backtrack_nodeID}) && $node->get_type() ne "terminal_right_exon") { + print $self->toString(); + print Dumper ($node); + confess "Error, node has undefined reverse backtrack ID but is not a terminal right exon: " . $node->toString() . "\n"; + } + } + + return; +} + + +#### +sub _extend_splice_paths_to_termini { + my $self = shift; + + my @splice_paths = $self->_get_valid_splice_paths(); + foreach my $splice_path (@splice_paths) { + my @ordered_nodeIDs = $splice_path->get_ordered_nodeIDs(); + + my $left_nodeID = $ordered_nodeIDs[0]; + my $right_nodeID = $ordered_nodeIDs[$#ordered_nodeIDs]; + + print "-extending splice_path [" . join (",", @ordered_nodeIDs) . "] maximally to the left and right:\n" if $SEE; + my @left_extension_nodes = $self->_extend_node_maximally_left($left_nodeID); + @left_extension_nodes = grep { $_ != $left_nodeID } @left_extension_nodes; # remove the nucleating node + + my @right_extension_nodes = $self->_extend_node_maximally_right($right_nodeID); + @right_extension_nodes = grep { $_ != $right_nodeID } @right_extension_nodes; + + unless (@left_extension_nodes && @right_extension_nodes) { + confess "Error, couldn't extend splice path to the left or right to include termini\n" + . "left: @left_extension_nodes\n" + . "right: @right_extension_nodes\n " + . $splice_path->toString(); + } + + ## create alignment assembly: + my @all_nodes_in_maximal_path = (@left_extension_nodes, @ordered_nodeIDs, @right_extension_nodes); + my $cdna_alignment = $self->_instantiate_cdna_alignment_from_nodeIDs(@all_nodes_in_maximal_path); + + $self->_add_alignment_assembly($cdna_alignment); + } + return; +} + +#### +sub _extend_node_maximally_left { + my $self = shift; + my (@nodeIDs) = @_; + my $highest_scoring_path = $self->_perform_traceback("forward_backtrack", \@nodeIDs); + my $highest_scoring_nodelist_aref = $highest_scoring_path->{path_nodeIDs_aref}; + return (@$highest_scoring_nodelist_aref); +} + + +#### +sub _extend_node_maximally_right { + my $self = shift; + my (@nodeIDs) = @_; + my $highest_scoring_path = $self->_perform_traceback("reverse_backtrack", \@nodeIDs); + my $highest_scoring_nodelist_aref = $highest_scoring_path->{path_nodeIDs_aref}; + return (@$highest_scoring_nodelist_aref); +} + + +#### +sub _perform_traceback { + my $self = shift; + my ($traceback_direction, $nodeIDs_aref) = @_; + + my @paths_and_score_structs; + + foreach my $nodeID (@$nodeIDs_aref) { + ## do backtracking from forward DP calculation: + my @nodes_in_path; + my $next_path_node = $nodeID; + my $path_score = 0; + while (defined $next_path_node) { + my $node_obj = $self->get_graph_node_via_nodeID($next_path_node); + my $ev_support = $node_obj->get_num_evidence_support(); + $path_score += $ev_support; + push (@nodes_in_path, $next_path_node); + if ($traceback_direction eq "forward_backtrack") { + $next_path_node = $node_obj->{forward_backtrack_nodeID}; + } + elsif ($traceback_direction eq "reverse_backtrack") { + $next_path_node = $node_obj->{reverse_backtrack_nodeID}; + } + else { + confess "Don't understand traceback direction: $traceback_direction "; + } + } + my $path_and_score_struct = { score => $path_score, + path_nodeIDs_aref => [@nodes_in_path], + }; + push (@paths_and_score_structs, $path_and_score_struct); + + } + + @paths_and_score_structs = reverse sort {$a->{score}<=>$b->{score}} @paths_and_score_structs; + + my $highest_scoring_path = shift @paths_and_score_structs; + + return ($highest_scoring_path); + +} + +=tree_traversal_replaced_by_DP_method + + +#### +my $path_no = 0; +sub _tree_traversal { + my $self = shift; + my ($node_obj, $traversal_direction, $current_path_score, $current_path_list_aref, + $max_score_sref, $max_path_href) = @_; + + #print $self->toString(); + #print $self->toAlignIllustration(60); + + # print "_tree_traversal() current_path: ". join (",", @$current_path_list_aref) . "\n" if $SEE; + + $self->_ensure_unique_path_nodes(@$current_path_list_aref); # defensive programming + + my $edge_key = "_" . $traversal_direction . "_nodeID_href"; + + my %edges = %{$node_obj->{$edge_key}}; + + ## add current node evidence score to the sum score + + + if (%edges) { + ## not a terminal node + ## explore each of the linked nodes: + my @linked_nodes = keys %edges; + foreach my $linked_node (@linked_nodes) { + my $linked_node_obj = $self->get_graph_node_via_nodeID($linked_node); + ## add linked node to path list: + my @current_path_list = @$current_path_list_aref; + push (@current_path_list, $linked_node); + my $num_evidence_support = $linked_node_obj->get_num_evidence_support(); + my $linked_node_path_score = $current_path_score + $num_evidence_support; + + $self->_tree_traversal($linked_node_obj, $traversal_direction, $linked_node_path_score, [@current_path_list], + $max_score_sref, $max_path_href); + } + + } + else { + ## a terminal node + ## base case! + if ($node_obj->get_type() !~ /terminal/) { + ## supposed to be a terminal node. + confess "Reached base case but not a terminal node! " . $node_obj->toString() . "\n" . $self->toString(); + } + + $path_no++; + print "$path_no _tree_traversal($traversal_direction) terminal_reached. PATH: " . join (",", @$current_path_list_aref) . "\n" if $SEE; + if ($current_path_score >= $$max_score_sref) { #includes case where incoming (starting) node is in fact a terminal node. + ## make this maximum scoring path: + $max_path_href->{score} = $current_path_score; + $max_path_href->{path_nodeIDs_aref} = [@$current_path_list_aref]; + $$max_score_sref = $current_path_score; + } + } + return; + +} + +=cut + +#### +sub _ensure_unique_path_nodes { + my $self = shift; + my @nodes = @_; + my %check; + foreach my $node (@nodes) { + if ($check{$node}) { + confess $self->toString() . "Error, path list: " . join (",", @nodes) . " has redundant node: $node\n"; + } + $check{$node} = 1; # log it! + } + return; # all good! +} + + +#### +sub _append_singletons_to_assembly_list { + my $self = shift; + foreach my $node_obj (@{$self->{_singleton_exons}}) { + my $nodeID = $node_obj->get_nodeID(); + my $cdna_alignment = $self->_instantiate_cdna_alignment_from_nodeIDs($nodeID); + $self->_add_alignment_assembly($cdna_alignment); + } + return; +} + + + +#### +sub _instantiate_cdna_alignment_from_nodeIDs { + my $self = shift; + my @nodeIDs = @_; + + my $spliced_orientation = undef; + + ## add alignment segments for all non-intron nodes: + my @coordsets; + foreach my $nodeID (@nodeIDs) { + my $node_obj = $self->get_graph_node_via_nodeID($nodeID); + my $node_type = $node_obj->get_type(); + if ($node_type eq "intron") { next; } ## no introns! + + my ($lend, $rend) = $node_obj->get_coords(); + my $orient = $node_obj->get_orient(); + my ($end5, $end3) = ($orient eq '+') ? ($lend, $rend) : ($rend, $lend); + push (@coordsets, [$end5, $end3]); + + unless (defined $spliced_orientation) { + $spliced_orientation = $orient; + } + + if ($spliced_orientation ne $orient) { + confess "Error, alignment segments of opposite orientation are in the same extended splice path! @nodeIDs\n" . $self->toString(); + } + } + + @coordsets = sort {$a->[0]<=>$b->[0]} @coordsets; + my $match_coord_sum = 0; + my @alignment_segments; + foreach my $coordset (@coordsets) { + my ($end5, $end3) = @$coordset; + my $seg_length = abs ($end3 - $end5) + 1; + my $match_lend = $match_coord_sum + 1; + my $match_rend = $match_coord_sum + $seg_length; + + my $alignment_segment = new CDNA::Alignment_segment($end5, $end3, $match_lend, $match_rend, 100); # 100% identity + + push (@alignment_segments, $alignment_segment); + + $match_coord_sum += $seg_length; + } + + my $cdna_alignment = new CDNA::CDNA_alignment($match_coord_sum, \@alignment_segments); + + $cdna_alignment->set_spliced_orientation($spliced_orientation); + $cdna_alignment->force_spliced_validation($spliced_orientation); + + ## hack in the node list as a hidden object attribute. + $cdna_alignment->{__Splice_graph_assembler_nodeID_list} = [@nodeIDs]; + + return ($cdna_alignment); +} + +#### +sub _correlate_assemblies_with_incoming_alignments { + my $self = shift; + my @assemblies = $self->get_assemblies(); + + my %alignment_accs; # keys to alignment objects + my %incorporated_alignments; + foreach my $assembly (@assemblies) { + + my $num_segments_in_assembly = $assembly->get_num_segments(); + my @incorporated_alignments; + + foreach my $alignment (@{$self->{_incoming_alignments}}) { + my $alignment_acc = $alignment->get_acc(); + + ## store alignment object in lookup table for later retrieval. + unless (exists $alignment_accs{$alignment_acc}) { + $alignment_accs{$alignment_acc} = $alignment; + } + + ## check for alignment incorporation: + if ($assembly->encapsulates($alignment, $FUZZ_DIST) && + ( + ## since assemblies have fixed spliced orient, compat will fail when encapsulation + ## of single exon alignments of ambiguous orientation exists + ($num_segments_in_assembly == 1 && $alignment->get_spliced_orientation() eq '?') + || + ## do rigorous compatibility check: + ($assembly->is_compatible($alignment, $FUZZ_DIST)) + ) + ) + { + push (@incorporated_alignments, $alignment_acc); + $incorporated_alignments{$alignment_acc} = 1; + } + } + $assembly->set_acc( join ("/", @incorporated_alignments)); # use / as DELIMETER for cdna accessions + } + + my @unincorporated_alignments; + foreach my $alignment_acc (keys %alignment_accs) { + unless ($incorporated_alignments{$alignment_acc}) { + push (@unincorporated_alignments, $alignment_accs{$alignment_acc}); + } + } + + + @{$self->{_unincorporated_alignments}} = (); #clear + + if (@unincorporated_alignments) { + @{$self->{_unincorporated_alignments}} = @unincorporated_alignments; + return (1); # still have unincorporated alignments! + } + + return(0); ## all alignments are accounted for. +} + +#### +sub _explore_assemblies_from_unincorporated_alignments { + my $self = shift; + print "method _explore_assemblies_from_unincorporated_alignments()\n" if $SEE; + ## decompose unincorporated alignments and create maximal paths that contain them. + + my @added_alignment_assemblies; + + + my @unincorporated_alignments = $self->get_unincorporated_alignments(); + + ## examine in order of descending numbers of segments and length: + @unincorporated_alignments = reverse sort { $a->{num_segments} <=> $b->{num_segments} + || + $a->{cdna_length} <=> $b->{cdna_length} } @unincorporated_alignments; + + + + foreach my $unincorporated_alignment (@unincorporated_alignments) { + print "\nUnincorporated alignment under scrutiny: " . $unincorporated_alignment->toToken() . "\n" if $SEE; + + my $orient = $unincorporated_alignment->get_spliced_orientation(); + my ($alignment_lend, $alignment_rend) = $unincorporated_alignment->get_coords(); + + my $found_in_new_assembly_flag = 0; + foreach my $new_assembly (@added_alignment_assemblies) { + if ($new_assembly->is_compatible($unincorporated_alignment, $FUZZ_DIST) && $new_assembly->encapsulates($unincorporated_alignment, $FUZZ_DIST)) { + $found_in_new_assembly_flag = 1; + last; + } + } + if ($found_in_new_assembly_flag) { + print "-found in a newly created assembly based on an earlier unincorporated alignment. Now accounted for.\n" if $SEE; + next; + } + + my $num_segments = $unincorporated_alignment->get_num_segments(); + + my @all_nodes; + + if ($num_segments == 1) { + ## must be included in some terminal exon not already represented by maximal splice paths (only explanation) + ## find a maximal path that includes this segment stemming from a terminal exon + + my @terminal_nodes = $self->_find_terminal_exons_encompassing_segment($alignment_lend, $alignment_rend, $orient); + unless (@terminal_nodes) { + confess "Error, searching for terminal nodes that encompass unincorproated singleton segment and none found.\n" + . $self->toString(); + } + ## find maximal path: + my @paths_from_terminal_node; + foreach my $terminal_node (@terminal_nodes) { + my $nodeID = $terminal_node->get_nodeID(); + my $type = $terminal_node->get_type(); + my $max_path_href; + if ($type eq "terminal_left_exon") { + ## must extend to the right + $max_path_href = $self->_perform_traceback("forward_backtrack", [$nodeID]); #_extend_node_maximally_right($nodeID); + } + elsif ($type eq "terminal_right_exon") { + $max_path_href = $self->_perform_traceback("reverse_backtrack", [$nodeID]); #_extend_node_maximally_left($nodeID); + } + else { + confess "Error, terminal node is supposed to be left or right, but type($type) is not recognized."; + } + + push (@paths_from_terminal_node, $max_path_href); + } + ## get highest scoring path: + @paths_from_terminal_node = reverse sort { $a->{score}<=>$b->{score} } @paths_from_terminal_node; + my $highest_scoring_path = shift @paths_from_terminal_node; + my $node_IDs_in_path_aref = $highest_scoring_path->{path_nodeIDs_aref}; + @all_nodes = @$node_IDs_in_path_aref; + } + + else { + ## has an intron! Some alternate termini needed that were not found in a maximal splice path + + my @segments = $unincorporated_alignment->get_alignment_segments(); + my $first_segment = shift @segments; + my $last_segment = pop @segments; + + ## create node list for internal segments and introns: + my @nodes_in_central_path; + my @introns = $unincorporated_alignment->get_intron_coords(); + foreach my $intron (@introns) { + my ($lend, $rend) = @$intron; + my $node_obj = $self->_get_graph_node_via_coords_n_type("intron", $lend, $rend, $orient); + my $nodeID = $node_obj->get_nodeID(); + push (@nodes_in_central_path, $nodeID); + } + foreach my $internal_segment (@segments) { + my ($lend, $rend) = $internal_segment->get_coords(); + my $node_obj = $self->_get_graph_node_via_coords_n_type("internal_exon", $lend, $rend, $orient); + my $nodeID = $node_obj->get_nodeID(); + push (@nodes_in_central_path, $nodeID); + } + my ($first_segment_lend, $first_segment_rend) = $first_segment->get_coords(); + my @candidate_preceding_nodes = $self->_find_candidate_preceding_exons($first_segment_lend, $first_segment_rend, $orient); + unless (@candidate_preceding_nodes) { + confess "Error, no candidate preceding nodes for ($first_segment_lend, $first_segment_rend, $orient)"; + } + + my ($last_segment_lend, $last_segment_rend) = $last_segment->get_coords(); + my @candidate_following_nodes = $self->_find_candidate_following_exons($last_segment_lend, $last_segment_rend, $orient); + unless (@candidate_following_nodes) { + confess "Error, no candidate following nodes for ($last_segment_lend, $last_segment_rend, $orient)"; + } + + ## find maximal paths left and right: + my @nodeIDs_left; + foreach my $candidate_preceding_node (@candidate_preceding_nodes) { + print "Candidate preceding node: " . $candidate_preceding_node->toString() . "\n" if $SEE; + my $nodeID = $candidate_preceding_node->get_nodeID(); + push (@nodeIDs_left, $nodeID); + } + my @path_left = $self->_extend_node_maximally_left(@nodeIDs_left); + + my @nodeIDs_right; + foreach my $candidate_following_node (@candidate_following_nodes) { + print "Candidate following node: " . $candidate_following_node->toString() . "\n" if $SEE; + my $nodeID = $candidate_following_node->get_nodeID(); + push (@nodeIDs_right, $nodeID); + } + my @path_right = $self->_extend_node_maximally_right(@nodeIDs_right); + + @all_nodes = (@path_left, @nodes_in_central_path, @path_right); + } + + + my $assembly = $self->_instantiate_cdna_alignment_from_nodeIDs(@all_nodes); + + ## verify that this assembly accounts for the unincorporated alignment: + unless ( (my $compatible_result = $assembly->is_compatible($unincorporated_alignment, $FUZZ_DIST)) + && + (my $encapsulated_result = $assembly->encapsulates($unincorporated_alignment, $FUZZ_DIST)) ) { + confess "Error, created new assembly to house missing alignment, but it doesn't afterall!\n" + . "unincorp: " . $unincorporated_alignment->toToken() . "\n" + . "newAsmb: " . $assembly->toToken() . "\n" + . "compatible: $compatible_result\n" + . "encapsulated: $encapsulated_result\n"; + } + print "-verified incorporation in new assembly.\n" if $SEE; + push (@added_alignment_assemblies, $assembly); + } + + ## add to assembly list + $self->_add_alignment_assembly(@added_alignment_assemblies); + + return; +} + +#### +sub _find_candidate_preceding_exons { + my $self = shift; + my ($lend, $rend, $orient) = @_; + + ## must have same orient, share rend boundary, and be internal or terminal exons that contain it + my @candidate_exons; + foreach my $exon (@{$self->{_internal_exons}}, @{$self->{_terminal_left_exons}}) { + my ($exon_lend, $exon_rend) = $exon->get_coords(); + my $exon_orient = $exon->get_orient(); + if ($exon_orient eq $orient && $lend + $FUZZ_DIST >= $exon_lend && $exon_rend == $rend) { + push (@candidate_exons, $exon); + } + } + + return (@candidate_exons); +} + +#### +#### +sub _find_candidate_following_exons { + my $self = shift; + my ($lend, $rend, $orient) = @_; + + ## must have same orient, share rend boundary, and be internal or terminal exons that contain it + my @candidate_exons; + foreach my $exon (@{$self->{_internal_exons}}, @{$self->{_terminal_right_exons}}) { + my ($exon_lend, $exon_rend) = $exon->get_coords(); + my $exon_orient = $exon->get_orient(); + if ($exon_orient eq $orient && $rend - $FUZZ_DIST <= $exon_rend && $exon_lend == $lend) { + push (@candidate_exons, $exon); + } + } + + return (@candidate_exons); +} + + + + + + + +###################################################### +###################################################### + +package Splice_graph_node; + +use strict; +use warnings; +use Carp; + +my $nodeID = 0; ## class attribute + +#### +sub new { + my $packagename = shift; + + my ($type, $lend, $rend, $orient) = @_; + + ## type can be: intron | internal_exon | terminal_left_exon | terminal_right_exon | singleton_exon + $nodeID++; + + # object atts: + my $self = { + nodeID => $nodeID, + type => undef, + lend => undef, + rend => undef, + orient => undef, + length => undef, + num_evidence_support => 1, # indicate number of transcripts supporting this feature + + ## graph connections: + _prev_nodeID_href => {}, ## keys nodeIDs connected to before this node + _next_nodeID_href => {}, ## keys nodeIDs connected to after this node + + ## attributes for dynamic programming calculations to find longest path + forward_base_score => 0, ## calculation performed from left to right, with right to left backtracking + forward_path_score => 0, + forward_backtrack_nodeID => undef, + + reverse_base_score => 0, ## calculation performed from right to left with backtracking from left to right + reverse_path_score => 0, + reverse_backtrack_nodeID => undef, + + }; + + bless ($self, $packagename); + + ## use set methods to set attribute values and validate settings: + $self->set_coords($lend, $rend); + $self->set_type($type); + $self->set_orient($orient); + + return ($self); +} + +#### +sub set_num_evidence_support { + my $self = shift; + my ($evidence_support) = @_; + + $self->{num_evidence_support} = $evidence_support; + return; +} + +#### +sub increment_evidence_support { + my $self = shift; + my ($inc_val) = @_; + if (defined $inc_val) { + $self->{num_evidence_support} += $inc_val; + } + else { + ## just add one. + $self->{num_evidence_support}++; + } + return; +} + +#### +sub get_num_evidence_support { + my $self = shift; + return ($self->{num_evidence_support}); +} + +#### +sub toString { + my $self = shift; + my $text = "node: " . $self->get_nodeID() + . "\t" . join ("-", $self->get_coords()) + . "\torient(" . $self->get_orient() . ")" + . "\t" . $self->get_type() + . "\tEvSupport: " . $self->get_num_evidence_support(); + + return ($text); +} + +#### +sub get_nodeID { + my $self = shift; + return ($self->{nodeID}); +} + + +#### +sub set_coords { + my $self = shift; + my ($lend, $rend) = @_; + + unless ($lend =~ /^\d+$/ && $rend =~ /^\d+$/) { + confess "Error, coordinates [$lend] or [$rend] are not integers"; + } + + unless ($lend <= $rend) { + confess "Error, coordinates out of order"; + } + + $self->{lend} = $lend; + $self->{rend} = $rend; + + $self->{length} = $rend - $lend + 1; + + return; +} + +#### +sub get_coords { + my $self = shift; + return ($self->{lend}, $self->{rend}); +} + +#### +sub set_type { + my $self = shift; + my ($type) = @_; + unless ($type =~ /^(intron|internal_exon|terminal_left_exon|terminal_right_exon|singleton_exon)$/) { + confess "type $type is not acceptible"; + } + + $self->{type} = $type; + return; +} + +#### +sub get_type { + my $self = shift; + return ($self->{type}); +} + +#### +sub set_orient { + my $self = shift; + my ($orient) = @_; + unless ($orient =~ /^[\+\-]$/) { + confess "Error, orient($orient) is not allowed here"; + } + + $self->{orient} = $orient; + return; +} + +#### +sub get_orient { + my $self = shift; + return ($self->{orient}); +} + + +#### +sub connect_this_prev_nodeID { + my $self = shift; + my ($nodeID) = @_; + $self->{_prev_nodeID_href}->{$nodeID} = 1; + return; +} + +#### +sub connect_this_next_nodeID { + my $self = shift; + my ($nodeID) = @_; + $self->{_next_nodeID_href}->{$nodeID} = 1; + return; +} + +#### +sub get_connected_prev_nodeIDs { + my $self = shift; + my @prev_nodeIDs = keys %{$self->{_prev_nodeID_href}}; + return (@prev_nodeIDs); +} + +#### +sub get_connected_next_nodeIDs { + my $self = shift; + my @next_nodeIDs = keys %{$self->{_next_nodeID_href}}; + return (@next_nodeIDs); +} + +#### +sub has_next_nodeID { + my $self = shift; + my ($nodeID) = @_; + if ($self->{_next_nodeID_href}->{$nodeID}) { + return (1); #yes + } + else { + return (0); # no + } +} + +#### +sub has_prev_nodeID { + my $self = shift; + my ($nodeID) = @_; + if ($self->{_prev_nodeID_href}->{$nodeID}) { + return (1); #yes + } + else { + return (0); #no + } +} + +############################################# +############################################# + +package Splice_graph_path; + +use strict; +use warnings; +use Carp; + +my $PATH_ID = 0; ## class variable + +#### +sub new { + my $packagename = shift; + + my ($lend, $rend, $orient, $ordered_nodeIDs_aref) = @_; + + unless ($lend =~ /^\d+$/ && $rend =~ /^\d+$/) { + confess "Error, lend $lend and rend $rend must be integers"; + } + unless ($orient =~ /^[\+\-]$/) { + confess "Error, orient($orient) is invalid"; + } + + unless (@$ordered_nodeIDs_aref) { + confess "Error, need list of ordered nodeIDs as constructor param"; + } + + $PATH_ID++; + + my $self = { + _lend => $lend, + _rend => $rend, + _orient => $orient, + _ordered_nodeIDs => [@$ordered_nodeIDs_aref], # a path is an ordered list of intron and exon nodeIDs traversed + _pathID => $PATH_ID, + + }; + + bless ($self, $packagename); + + return ($self); +} + +#### +sub get_ordered_nodeIDs { + my $self = shift; + return (@{$self->{_ordered_nodeIDs}}); +} + +#### +sub toString { + my $self = shift; + my @ordered_nodeIDs = $self->get_ordered_nodeIDs(); + return (join (",", @ordered_nodeIDs)); +} + +#### +sub is_subpath_of { + my $self = shift; + my ($other_path) = @_; + + my $self_path_string = "," . $self->toString() . ","; ## add comma terminal delimiters + + my $other_path_string = "," . $other_path->toString() . ","; + + if ($other_path_string =~ /$self_path_string/) { + return (1); #true + } + else { + return (0); #false + } +} + +#### +sub is_compatible () { + my $self = shift; + my ($other_splice_path_obj) = @_; + + ## make sure they have at least one node in common: + my @node_IDs_A = $self->get_ordered_nodeIDs(); + my @node_IDs_B = $other_splice_path_obj->get_ordered_nodeIDs(); + + my $found_nodeID_in_common = 0; + my $common_nodeID = undef; + { # do this in a more restricted scope + my %nodeIDs; + foreach my $nodeID (@node_IDs_A) { + $nodeIDs{$nodeID} = 1; + } + foreach my $nodeID (@node_IDs_B) { + if ($nodeIDs{$nodeID}) { + $found_nodeID_in_common = 1; + $common_nodeID = $nodeID; + last; + } + } + } + unless ($found_nodeID_in_common) { + return (0); # false; no common node so definitely incompatible + } + + ## examine all nodes adjacent to common node; they should be completely identical + my $common_node_pos_A = $self->_find_ele_index_pos_via_value(\@node_IDs_A, $common_nodeID); + my $common_node_pos_B = $self->_find_ele_index_pos_via_value(\@node_IDs_B, $common_nodeID); + + ## search left: + my ($i, $j); + for ($i=$common_node_pos_A, $j=$common_node_pos_B; $i >= 0 && $j >= 0; $i--, $j--) { + my $A_node = $node_IDs_A[$i]; + my $B_node = $node_IDs_B[$j]; + if ($A_node ne $B_node) { + # incompatible + return (0); + } + } + + ## search right: + for ($i = $common_node_pos_A, $j = $common_node_pos_B; $i <= $#node_IDs_A && $j <= $#node_IDs_B; $i++,$j++) { + my $A_node = $node_IDs_A[$i]; + my $B_node = $node_IDs_B[$j]; + if ($A_node ne $B_node) { + # incompatible + return (0); + } + } + + ## if got here, then passed compatibility tests. + return (1); # Yes, compatible + +} + + +sub _find_ele_index_pos_via_value { + my $self = shift; + my ($list_aref, $value) = @_; + for (my $i = 0; $i <= $#$list_aref; $i++) { + if ($list_aref->[$i] eq $value) { + return ($i); + } + } + confess "Error finding index position of $value in list"; +} + +#### +sub get_pathID { + my $self = shift; + return ($self->{_pathID}); +} + +#### +sub get_coords { + my $self = shift; + return ($self->{_lend}, $self->{_rend}); +} + +#### +sub get_orient { + my $self = shift; + return ($self->{_orient}); +} + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/CIGAR.pm b/99.scripts/trinity_utils/PerlLib/CIGAR.pm new file mode 100644 index 0000000..e84aaa8 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/CIGAR.pm @@ -0,0 +1,94 @@ +package CIGAR; + +use strict; +use warnings; + +use Nuc_translator; + +#### +sub construct_cigar { + my ($genome_coords_aref, $query_coords_aref, $read_length, # required + $genome_seq_sref, $strand # optional + ) = @_; + + my $cigar = ""; + + + for (my $i = 0; $i <= $#$genome_coords_aref; $i++) { + + my ($curr_genome_lend, $curr_genome_rend) = @{$genome_coords_aref->[$i]}; + my ($curr_query_lend, $curr_query_rend) = @{$query_coords_aref->[$i]}; + + if ($i == 0) { + + if ($curr_query_lend > 1) { + $cigar .= ($curr_query_lend - 1) . "S"; + } + } + else { + my ($prev_genome_lend, $prev_genome_rend) = @{$genome_coords_aref->[$i-1]}; + my ($prev_query_lend, $prev_query_rend) = @{$query_coords_aref->[$i-1]}; + + if ( (my $delta_genome = $curr_genome_lend - $prev_genome_rend) > 1) { + my $deletion_intron_char = ($genome_seq_sref && $strand) ? &_check_intron_consensus($prev_genome_rend, $curr_genome_lend, $genome_seq_sref, $strand) : 'D'; # intron or deletion? + $cigar .= ($delta_genome-1) . $deletion_intron_char; + } + if ( (my $delta_query = $curr_query_lend - $prev_query_rend) > 1) { + $cigar .= ($delta_query-1) . "I"; + } + } + + + my $len = $curr_genome_rend - $curr_genome_lend + 1; + + $cigar .= "$len" . "M"; + + if ($i == $#$genome_coords_aref) { + + if ($curr_query_rend < $read_length) { + $cigar .= ($read_length - $curr_query_rend) . "S"; + } + } + + } + + return($cigar); +} + + +#### +sub _check_intron_consensus { + my ($left_segment_bound, $right_segment_bound, $genome_sref, $strand) = @_; + + my $left_dinuc = uc substr($$genome_sref, $left_segment_bound, 2); + + my $right_dinuc = uc substr($$genome_sref, $right_segment_bound-3, 2); + + if ($strand eq '-') { + $left_dinuc = &reverse_complement($right_dinuc); + $right_dinuc = &reverse_complement($left_dinuc); + } + + if ( + ( ($left_dinuc eq "GT" || $left_dinuc eq "GC") && $right_dinuc eq "AG") + || + ($left_dinuc eq "CT" && $right_dinuc eq "AC") + ) { + + return("N"); # got splice pair + } + else { + return("D"); # deletion + } + + +} + + + + + + + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/CMD_processor.pm b/99.scripts/trinity_utils/PerlLib/CMD_processor.pm new file mode 100644 index 0000000..3bef54c --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/CMD_processor.pm @@ -0,0 +1,63 @@ +package CMD_processor; + +use strict; +use warnings; + +use Carp; + +use vars qw (@ISA @EXPORT); ## set in script using this module for verbose output. +@ISA = qw(Exporter); +@EXPORT = qw(process_cmd process_parallel_cmds); + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret ($ret)"; + } + + return; # all good. +} + +#### +sub process_parallel_cmds { + my (@cmds) = @_; + + print "Processing commands:\n" . join("\n", @cmds) . "\n\n"; + + + foreach my $cmd (@cmds) { + my $pid = fork(); + unless ($pid) { + # child + &process_cmd($cmd); + exit(0); + } + } + + my $process_failed = 0; + while (my $pid = wait()) { + if ($pid < 0) { + last; + } + else { + if ($?) { + print STDERR "-process $pid exited ($?) indicating error.\n"; + $process_failed = 1; + } + } + } + + if ($process_failed) { + die "Error, at least one command failed. See above errors for more details. "; + } + + return; # all good. +} + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/COMMON.pm b/99.scripts/trinity_utils/PerlLib/COMMON.pm new file mode 100644 index 0000000..ff5a394 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/COMMON.pm @@ -0,0 +1,31 @@ +package COMMON; + +use strict; +use warnings; +use Carp; + +$ENV{LC_ALL} = 'C'; # needed for sorting order. + +#### +sub get_sort_exec { + my ($num_threads) = @_; + + # check it like so: + # perl -MCOMMON -e 'print COMMON::get_sort_exec(4);' + + my $sort_exec = `sh -c "command -v sort"`; + unless ($sort_exec =~ /\w/) { + confess "Error, cannot find sort utility"; + } + $sort_exec =~ s/\s//g; + + my $help_text = `$sort_exec --help`; + if ($help_text =~ m|--parallel|) { + ## could do simple versioning check, but I don't remember which version started using parallel + $sort_exec = "$sort_exec --parallel=$num_threads"; + } + + return($sort_exec); +} + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/CanvasXpress/Heatmap.pm b/99.scripts/trinity_utils/PerlLib/CanvasXpress/Heatmap.pm new file mode 100644 index 0000000..45e9370 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/CanvasXpress/Heatmap.pm @@ -0,0 +1,164 @@ +package CanvasXpress::Heatmap; + +use strict; +use warnings; + + +sub draw { + my %inputs = @_; + + # structure of input hash: + # + # %inputs = ( samples => [ 'sampleA', 'sampleB', 'sampleC', ...], + # value_matrix => [ ['featureA', valA, valB, valC, ...], + # ['featureB', valA, valB, valC, ...], ... , + # ], + # ## and optionally: + # feature_tree => "", # string containing the newick formatted tree + # sample_tree => "", # ditto above + # feature_descriptions => ['name of feature A', 'name of feature B', ...], + # cluster_features => 0|1, ## clustering done in browser, exclusive with feature_tree option + # cluster_samples => 0|1, + # ); + # + + + + + + my $html = "\n"; + $html .= "\n"; + + + $html .= " + + +
+ +
+ + +__EOJS__ + + ; + + return($html); +} + + +1; #EOM + + diff --git a/99.scripts/trinity_utils/PerlLib/ColorGradient.pm b/99.scripts/trinity_utils/PerlLib/ColorGradient.pm new file mode 100644 index 0000000..b91ab7b --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/ColorGradient.pm @@ -0,0 +1,104 @@ +package ColorGradient; + +use strict; +use warnings; + + +# RGB values off (0), on (1), increasing (I), or decreasing (D). +my @ColorPhases = ( ['1', 'I', '0'], + ['D', '1', '0'], + ['0', '1', 'I'], + ['0', 'D', '1'], + ['I', '0', '1'], + ); + +my $num_color_phases = scalar @ColorPhases; +my $discrete_color_phase_percent = 1 /$num_color_phases * 100; + +#### +sub get_RGB_gradient { + my ($num_colors) = @_; + + ## Returns a list of [R,G,B], ... values. + + unless ($num_colors =~ /^\d+$/) { + die "Error, $num_colors should be a number"; + } + + my @colors; + $num_colors--; + for my $color_entry (0..$num_colors) { + + my $percentage = ($color_entry-0.000001) / $num_colors * 100; # don't ever want to reach 100% + + my $index = int ($percentage / $discrete_color_phase_percent); + + my $color_phase = $ColorPhases[$index]; + + my $ratio_into_phase = ($percentage - $discrete_color_phase_percent*$index) / $discrete_color_phase_percent; + + my $rgb_color = &_get_color($color_phase, $ratio_into_phase); + + push (@colors, $rgb_color); + } + + return (@colors); +} + +### +sub convert_RGB_hex { + my @rgb_aref_vals = @_; + + my @hex_vals; + foreach my $rgb_aref (@rgb_aref_vals) { + my ($r, $g, $b) = @$rgb_aref; + + my $hex_r = sprintf("%02x", $r); + my $hex_g = sprintf("%02x", $g); + my $hex_b = sprintf ("%02x", $b); + + push (@hex_vals, '#' . $hex_r . $hex_g . $hex_b); + } + return (@hex_vals); +} + +#### +sub _get_color { + my ($color_phase, $ratio_into_phase) = @_; + + my $variable_rgb_val = int($ratio_into_phase * 255 + 4/9); + + my @phase_vals = @$color_phase; + + + my @ret_rgb; + foreach my $phase_val (@phase_vals) { + my $rgb_val; + if ($phase_val eq '1') { + $rgb_val = 255; + } + elsif ($phase_val eq '0') { + $rgb_val = 0; + } + elsif ($phase_val eq 'I') { + $rgb_val = $variable_rgb_val; + } + elsif ($phase_val eq 'D') { + $rgb_val = 255 - $variable_rgb_val; + } + else { + die "Error, unrecognized phase value of $phase_val"; + } + + push (@ret_rgb, $rgb_val); + } + + return ([@ret_rgb]); +} + + + + +1; + + diff --git a/99.scripts/trinity_utils/PerlLib/DelimParser.pm b/99.scripts/trinity_utils/PerlLib/DelimParser.pm new file mode 100644 index 0000000..b5684a2 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/DelimParser.pm @@ -0,0 +1,291 @@ +#!/usr/bin/env perl + +# classes for DelimParser::Reader and DelimParser::Writer + +package DelimParser; +use strict; +use warnings; +use Carp; + +#### +sub new { + my ($packagename, $fh, $delimiter) = @_; + + unless ($fh && $delimiter) { + confess "Error, need filehandle and delimiter params"; + } + + + my $self = { _delim => $delimiter, + _fh => $fh, + + # set below in _init() + _column_headers => [], + }; + + + bless ($self, $packagename); + + return($self); +} + + +#### +sub get_fh { + my $self = shift; + return($self->{_fh}); +} + +#### +sub get_delim { + my $self = shift; + return($self->{_delim}); +} + +#### +sub get_column_headers { + my $self = shift; + return(@{$self->{_column_headers}}); +} + +#### +sub set_column_headers { + my $self = shift; + my (@columns) = @_; + + $self->{_column_headers} = \@columns; + + return; +} + +#### +sub get_num_columns { + my $self = shift; + return(length($self->get_column_headers())); +} + + +### +sub reconstruct_header_line { + my $self = shift; + my @column_headers = $self->get_column_headers(); + + my $header_line = join("\t", @column_headers); + return($header_line); +} + +### +sub reconstruct_line_from_row { + my $self = shift; + my $row_href = shift; + unless ($row_href && ref $row_href) { + confess "Error, must set row_href as param"; + } + + my @column_headers = $self->get_column_headers(); + + my @vals; + foreach my $col_header (@column_headers) { + my $val = $row_href->{$col_header}; + push (@vals, $val); + } + + my $row_text = join("\t", @vals); + + return($row_text); + +} + + +################################################## +package DelimParser::Reader; +use strict; +use warnings; +use Carp; +use Data::Dumper; + +our @ISA; +push (@ISA, 'DelimParser'); + +sub new { + my ($packagename) = shift; + my $self = $packagename->DelimParser::new(@_); + + $self->_init(); + + return($self); +} + + +#### +sub _init { + my $self = shift; + + my $fh = $self->get_fh(); + my $delim = $self->get_delim(); + + my $header_row = <$fh>; + chomp $header_row; + + unless ($header_row) { + confess "Error, no header row read."; + } + + my @fields = split(/$delim/, $header_row); + + $self->set_column_headers(@fields); + + + return; +} + +#### +sub get_row { + my $self = shift; + + my $fh = $self->get_fh(); + my $line = <$fh>; + unless ($line) { + return(undef); # eof + } + + my $delim = $self->get_delim(); + my @fields = split(/$delim/, $line); + chomp $fields[$#fields]; ## it's important that this is done after the delimiter splitting in case the last field is actually empty. + + + my @column_headers = $self->get_column_headers(); + + my $num_col = scalar (@column_headers); + my $num_fields = scalar(@fields); + + if ($num_col != $num_fields) { + confess "Error, line: [$line] " . Dumper(\@fields) . " is lacking $num_col fields: " . Dumper(\@column_headers); + } + + my %dict; + foreach my $colname (@column_headers) { + my $field = shift @fields; + $dict{$colname} = $field; + } + + return(\%dict); +} + + +#### +sub get_row_val { + my ($self, $row_href, $key) = @_; + + if (! exists $row_href->{$key}) { + confess "Error, row: " . Dumper($row_href) . " doesn't include key: [$key]"; + } + + return($row_href->{$key}); +} + + + +################################################## + +package DelimParser::Writer; +use strict; +use warnings; +use Carp; + +our @ISA; +push (@ISA, 'DelimParser'); + +sub new { + my ($packagename) = shift; + my ($ofh, $delim, $column_fields_aref, $FLAGS) = @_; + + ## FLAGS can be: + # NO_WRITE_HEADER|... + + unless (ref $column_fields_aref eq 'ARRAY') { + confess "Error, need constructor params: ofh, delim, column_fields_aref"; + } + + my $self = $packagename->DelimParser::new($ofh, $delim); + + $self->_initialize($column_fields_aref, $FLAGS); + + return($self); +} + + +#### +sub _initialize { + my $self = shift; + my $column_fields_aref = shift; + my $FLAGS = shift; + + unless (ref $column_fields_aref eq 'ARRAY') { + confess "Error, require column_fields_aref as param"; + } + + + my $ofh = $self->get_fh(); + my $delim = $self->get_delim(); + + + + $self->set_column_headers(@$column_fields_aref); + + unless (defined($FLAGS) && $FLAGS =~ /NO_WRITE_HEADER/) { + my $output_line = join($delim, @$column_fields_aref); + print $ofh "$output_line\n"; + } + + + return; +} + + +#### +sub write_row { + my $self = shift; + my $dict_href = shift; + + unless (ref $dict_href eq "HASH") { + confess "Error, need dict_href as param"; + } + + my $num_dict_fields = scalar(keys %$dict_href); + + my @column_headers = $self->get_column_headers(); + + + my $delim = $self->get_delim(); + + my @out_fields; + for my $column_header (@column_headers) { + my $field = $dict_href->{$column_header}; + unless (defined $field) { + confess "Error, missing value for required column field: $column_header"; + } + if ($field =~ /$delim/) { + # don't allow any delimiters to contaminate the field value, otherwise it'll introduce offsets. + $field =~ s/$delim/ /g; + } + # also avoid newlines, which will also break the output formatting. + if ($field =~ /\n/) { + $field =~ s/\n/ /g; + } + + push (@out_fields, $field); + } + + my $outline = join("\t", @out_fields); + + my $ofh = $self->get_fh(); + + print $ofh "$outline\n"; + + return; +} + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/EM.pm b/99.scripts/trinity_utils/PerlLib/EM.pm new file mode 100644 index 0000000..0bc6549 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/EM.pm @@ -0,0 +1,516 @@ +package EM; + +use strict; +use warnings; +use SAM_entry; +use Carp; + +use vars qw($TOKEN); +BEGIN { + $TOKEN = "$;"; +} + +my $MIN_EFF_TRANS_LENGTH = 10; # // large enough to keep abundance estimates from going absurdly high for short transcripts. + +#### +sub new { + my $packagename = shift; + my ($transcript_seqs_href, $fragment_length) = @_; + + unless ($fragment_length) { + $fragment_length = 1; # negligible with respect to the length of the transcript + } + + + # transcript_seqs_href: ( seq_acc => sequence, ...) + # trans_reads_aref: ( [trans_accA, read_accA], [trans_accB, read_accB], ...) + + unless ($transcript_seqs_href) { + confess "Error, need param: transcript_seqs_href"; + } + + + + my $self = { + + # mapping info + _trans_to_multi_map_counts => {}, # trans => { transA\$;transB\$;transC... } => count_of_reads + + # model params + _ENt => {}, # trans => expected read count + _theta => {}, # trans => fraction of all reads + + # transcript info + _trans_lengths => {}, # trans => length + _frag_length => $fragment_length, + + + # misc + _total_reads => 0, # total read count + + }; + + + bless ($self, $packagename); + + $self->_init_trans_lengths($transcript_seqs_href); + + return($self); + +} + + +#### +sub _init_trans_lengths { + my $self = shift; + my ($transcript_seqs_href) = @_; + + foreach my $transcript (keys %$transcript_seqs_href) { + + my $sequence = $transcript_seqs_href->{$transcript} or confess "Error, no sequence for transcript: $transcript"; + + $self->{_trans_lengths}->{$transcript} = length($sequence); + + } + + return; +} + +sub run { + my $self = shift; + my (%settings) = @_; + + my $max_iterations = $settings{max_iterations} || 1000; + my $min_delta_ML = $settings{min_delta_ML} || 0.01; + my $verbose_flag = $settings{verbose} || 0; + + + $self->_init_theta(); + + ######################### + ## Do the EM: + + my $prev_ML_val = undef; + + my $round = 0; + my $delta = 1; + + print STDERR "\tEM started\n"; + while (1) { + + $round++; + + $self->_E_step(); + + $self->_M_step(); + + + my $ml = $self->_compute_Max_Likelihood(); + + if (defined $prev_ML_val) { + $delta = $ml - $prev_ML_val; + } + + print STDERR "\rML[$round]: $ml $delta "; + + $prev_ML_val = $ml; + + if ($verbose_flag) { + print "\n\nEM_results, round: $round:\n"; + $self->report_results(); + print "\n"; + } + + if ($round >= $max_iterations || $delta < $min_delta_ML) { + + last; + } + + } + print STDERR "\n\tEM finished.\n"; + + return; +} + + +#### +sub _init_theta { + my $self = shift; + + ########################### + ## init theta: + ## assume mutliply mapped reads equally divided among their mapped locations for init + + my @transcripts = $self->_get_all_transcripts(); + + my $total_reads = $self->{_total_reads}; + + foreach my $transcript (@transcripts) { + + my $read_count_sum = 0; + + my @combos = $self->_get_multi_map_list_including_transcript($transcript); + + foreach my $combo (@combos) { + + my @other_trans = $self->_get_transcripts_in_combo($combo); + + my $num_other = scalar(@other_trans); + my $count = $self->_get_multi_map_read_count_for_combo($transcript, $combo); + + $read_count_sum += ($count/$num_other); + } + + $self->{_theta}->{$transcript} = $read_count_sum / $total_reads; + + } + + return; +} + +#### +sub _get_all_transcripts { + my $self = shift; + + my @transcripts = keys %{$self->{_trans_to_multi_map_counts}}; + + return(@transcripts); +} + + +#### +sub _get_reads_mapped_to_transcript { + my $self = shift; + my ($transcript) = @_; + + if (exists $self->{_trans_to_reads_href}->{$transcript}) { + my @reads = @{$self->{_trans_to_reads_href}->{$transcript}}; + return(@reads); + } + else { + confess "Error, no reads mapped to transcript $transcript"; + } +} + +#### +sub _get_transcript_length { + my $self = shift; + my ($transcript) = @_; + + return($self->{_trans_lengths}->{$transcript}); +} + + + + +#### +sub _E_step { + my $self = shift; + + my @transcripts = $self->_get_all_transcripts(); + + ## do the E + foreach my $transcript (@transcripts) { + + my $expected_read_count = 0; + + + + my $trans_length = $self->_get_transcript_length($transcript) or die "Error, no length for transcript: $transcript"; + my $eff_trans_length = $trans_length - $self->{_frag_length} + 1; + if ($eff_trans_length < $MIN_EFF_TRANS_LENGTH) { + $eff_trans_length = $MIN_EFF_TRANS_LENGTH; + } + + + my $theta_t = $self->{_theta}->{$transcript}; + if (! defined $theta_t) { + die "Error, no theta($transcript), $theta_t"; + } + + + my @combos = $self->_get_multi_map_list_including_transcript($transcript); + + #print "Combos: @combos\n"; + + + foreach my $combo (@combos) { + + my $combo_count = $self->_get_multi_map_read_count_for_combo($transcript, $combo); + + #print "Got combo: $combo, count: $combo_count\n"; + + my @other_transcripts = split(/$TOKEN/, $combo); + if (scalar @other_transcripts == 1) { + # only current transcript: + unless ($other_transcripts[0] eq $transcript) { + confess "Error, mapped single transcript should be the current transcript"; + } + $expected_read_count += $combo_count; + } + else { + ## compute partial mapping. + + + my $numerator = (1 / $eff_trans_length) * $theta_t; + #my $numerator = $theta_t; + + + # demonator: sum for all t: P(r|t) * theta(t) + my $denominator = 0; + + + #print STDERR "Other transcripts: @other_transcripts\n"; + + foreach my $other_transcript (@other_transcripts) { # other transcripts includes the current transcript as well. + + my $other_trans_len = $self->_get_transcript_length($other_transcript); + unless ($other_trans_len) { + die "Error, no trans length for $other_transcript"; + } + + my $eff_other_trans_length = $other_trans_len - $self->{_frag_length} + 1; + if ($eff_other_trans_length < $MIN_EFF_TRANS_LENGTH) { + $eff_other_trans_length = $MIN_EFF_TRANS_LENGTH; + } + + my $other_theta = $self->{_theta}->{$other_transcript}; + + if (! defined $other_theta) { + die "no theta for $other_transcript, $other_theta"; + } + + my $val = (1 / $eff_other_trans_length) * $other_theta; + #my $val = $other_theta; + + $denominator += $val; + } + + #print "Fractional read count for $transcript = $numerator / $denominator\n"; + + my $fractional_read_count = $combo_count * ($numerator/$denominator); + + + $expected_read_count += $fractional_read_count; + } + } + + $self->{_ENt}->{$transcript} = $expected_read_count; + } + + return; +} + + +#### +sub _M_step { + my $self = shift; + + my $sum_NT = 0; + + foreach my $E (values %{$self->{_ENt}}) { + $sum_NT += $E; + } + + ## theta = ENt(t) / sum( for all t: ENt(t) ) + + + foreach my $transcript ($self->_get_all_transcripts()) { + my $new_theta = $self->{_ENt}->{$transcript} / $sum_NT; + + $self->{_theta}->{$transcript} = $new_theta; + } + + return; +} + + +#### +sub _compute_Max_Likelihood { + my $self = shift; + + # ML = sum:t (ENt * log(theta(t))) + + my $ML = 0; + + foreach my $transcript ($self->_get_all_transcripts()) { + + my $theta = $self->{_theta}->{$transcript}; + + if ($theta > 0) { + + my $ENt = $self->{_ENt}->{$transcript}; + + $ML += $ENt * log($theta); + + } + } + + return($ML); +} + + +#### +sub get_results { + my $self = shift; + + my $total_reads = $self->{_total_reads}; + + my @results; + + foreach my $transcript (sort $self->_get_all_transcripts()) { + + my $num_frags = $self->{_ENt}->{$transcript}; + my $len = $self->_get_transcript_length($transcript); + + $num_frags = sprintf("%.2f", $num_frags); + + my $fpkm = $num_frags / ($len/1e3) / ($total_reads/1e6); + $fpkm = sprintf("%.2f", $fpkm); + + my ($num_unique_reads, $num_multi_map_reads) = $self->count_reads_mapped_to_transcript($transcript); + + my $struct = { trans_id => $transcript, + length => $len, + unique_map => $num_unique_reads, + multi_map => $num_multi_map_reads, + expected_map => $num_frags, + FPKM => $fpkm, + }; + + push (@results, $struct); + + + + } + + + return (@results); +} + + +sub print_results { + my $self = shift; + + print $self->report_results(); + return; +} + + + +sub report_results { + my ($self) = shift; + + my @results = $self->get_results(); + + my $out_text = join("\t", "#transcript", "trans_length", "unique_map", "multi_map", "EM_frag_count", "FPKM") . "\n"; + + foreach my $result (@results) { + + $out_text .= join("\t", + $result->{trans_id}, + $result->{length}, + $result->{unique_map}, + $result->{multi_map}, + $result->{expected_map}, + $result->{FPKM}) . "\n"; + } + + $out_text .= "\n"; # last spacer + + return ($out_text); + +} + + + + + + +#### +sub _get_multi_map_list_including_transcript { + my $self = shift; + my ($transcript) = @_; + + my @combos = keys %{$self->{_trans_to_multi_map_counts}->{$transcript}}; + return(@combos); +} + +#### +sub _get_multi_map_read_count_for_combo { + my $self = shift; + my ($transcript, $combo) = @_; + + my $val = $self->{_trans_to_multi_map_counts}->{$transcript}->{$combo}; + + return($val); +} + +#### +sub _get_transcripts_in_combo { + my $self = shift; + my ($combo) = @_; + + my @transcripts = split(/$TOKEN/, $combo); + return(@transcripts); +} + + +#### +sub _create_combo { + my $self = shift; + my @transcripts = @_; + + my $combo = join($TOKEN, sort @transcripts); + + return($combo); +} + +#### +sub add_read { # really add_fragment: count pairs only once. + my $self = shift; + my @transcripts = @_; + + my $combo = $self->_create_combo(@transcripts); + + foreach my $transcript (@transcripts) { + $self->{_trans_to_multi_map_counts}->{$transcript}->{$combo}++; + } + + $self->{_total_reads}++; + + return; +} + +#### +sub count_reads_mapped_to_transcript { + my $self = shift; + my ($transcript) = @_; + + my $unique_read_count = 0; + my $multi_map_count = 0; + + + my @combos = $self->_get_multi_map_list_including_transcript($transcript); + + foreach my $combo (@combos) { + + my $combo_count = $self->_get_multi_map_read_count_for_combo($transcript, $combo); + + + my @other_transcripts = $self->_get_transcripts_in_combo($combo); + + if (scalar @other_transcripts == 1) { + # only current transcript: + $unique_read_count += $combo_count; + } + else { + $multi_map_count += $combo_count; + } + } + + return($unique_read_count, $multi_map_count); +} + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/Exons_to_geneobj.pm b/99.scripts/trinity_utils/PerlLib/Exons_to_geneobj.pm new file mode 100644 index 0000000..1861abb --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Exons_to_geneobj.pm @@ -0,0 +1,211 @@ +package main; +our $SEE; + +package Exons_to_geneobj; + +use strict; +use Longest_orf; +use Gene_obj; + +## No reason to instantiate. Use methods, fully qualified. +## allow for partial ORFS (missing start or stop codon in longest ORF) + +################### +## Public method ## +################### + +sub create_gene_obj { + my ($exons_href, $sequence_ref, $partial_info_href) = @_; + unless (ref $sequence_ref) { + die "Error, need reference to sequence as input parameter.\n"; + } + unless (ref $partial_info_href) { + $partial_info_href = {}; + } + + ## exons_ref should be end5's keyed to end3's for all exons. + my ($gene_struct_mod, $cdna_seq) = &get_cdna_seq ($exons_href, $sequence_ref); + + my $cdna_seq_length = length $cdna_seq; + my $long_orf_obj = new Longest_orf(); + + # establish long orf finding parameters. + $long_orf_obj->forward_strand_only(); + if ($partial_info_href->{"5prime"}) { + print "Exons_to_geneobj: Allowing 5' partials\n" if $SEE; + $long_orf_obj->allow_5prime_partials(); + } + $long_orf_obj->allow_3prime_partials(); ## Allow this by default. + + + $long_orf_obj->get_longest_orf($cdna_seq); + my ($end5, $end3) = $long_orf_obj->get_end5_end3(); + + print "CDS: $end5, $end3\n" if $SEE; + my $gene_obj = &create_gene ($gene_struct_mod, $end5, $end3); + + + ## Make adjustments if partial + my @exons = $gene_obj->get_exons(); + my $first_exon = $exons[0]; + my $last_exon = $exons[$#exons]; + + my $need_refine_flag = 0; + + if ($partial_info_href->{"5prime"}) { + $gene_obj->set_5prime_partial(1); + my $cds_exon = $first_exon->get_CDS_obj(); + if (ref $cds_exon) { + $cds_exon->{phase} = 0; + my $diff_coords = abs ($cds_exon->{end5} - $first_exon->{end5}); + if ($diff_coords > 0 && $diff_coords < 3) { + $cds_exon->{end5} = $first_exon->{end5}; + $need_refine_flag = 1; + + ## update phase of first exon + if ($diff_coords == 1) { + $cds_exon->{phase} = 2; + } elsif ($diff_coords == 2) { + $cds_exon->{phase} = 1; + } + + } + } + + } + + if ($partial_info_href->{"3prime"}) { + $gene_obj->set_3prime_partial(1); + my $cds_exon = $last_exon->get_CDS_obj(); + if (ref $cds_exon) { + my $diff_coords = abs ($cds_exon->{end3} - $last_exon->{end3}); + if ($diff_coords > 0 && $diff_coords < 3) { + $cds_exon->{end3} = $last_exon->{end3}; + $need_refine_flag = 1; + } + } + } + + ## Set the phase attribute for each cds exon: + my @cds_exons; + foreach my $exon ($gene_obj->get_exons()) { + if (my $cds = $exon->get_CDS_obj()) { + if (ref $cds) { + push (@cds_exons, $cds); + } + } + } + + if (@cds_exons) { + my $cds_length = 0; + my $first_cds = shift @cds_exons; + my $phase = $first_cds->{phase}; + unless ($phase) { + ## didn't set it for non 5' partials yet. + $phase = $first_cds->{phase} = 0; + } + $cds_length -= $phase; + $cds_length += $first_cds->length(); + foreach my $following_cds (@cds_exons) { + $following_cds->{phase} = $cds_length % 3; + $cds_length += $following_cds->length(); + } + } + + + if ($need_refine_flag) { + $gene_obj->refine_gene_object(); + } + + return ($gene_obj); +} + + +######################## +## Private methods ##### +######################## + +#### +sub get_cdna_seq { + my ($gene_struct, $assembly_seq_ref) = @_; + my (@end5s) = sort {$a<=>$b} keys %$gene_struct; + my $strand = "?"; + foreach my $end5 (@end5s) { + my $end3 = $gene_struct->{$end5}; + if ($end5 == $end3) { next;} + $strand = ($end5 < $end3) ? '+':'-'; + last; + } + if ($strand eq "?") { + print Dumper ($gene_struct); + die "ERROR: I can't determine what orientation the cDNA is in!\n"; + } + print NOTES "strand: $strand\n"; + my $cdna_seq; + my $gene_struct_mod = {strand=>$strand, + exons=>[]}; #ordered lend->rend coordinate listing. + foreach my $end5 (@end5s) { + #print $end5; + my $end3 = $gene_struct->{$end5}; + my ($coord1, $coord2) = sort {$a<=>$b} ($end5, $end3); + my $exon_seq = substr ($$assembly_seq_ref, $coord1 - 1, ($coord2 - $coord1 + 1)); + $cdna_seq .= $exon_seq; + push (@{$gene_struct_mod->{exons}}, [$coord1, $coord2]); + } + if ($strand eq '-') { + $cdna_seq = reverse_complement ($cdna_seq); + } + return ($gene_struct_mod, $cdna_seq); +} + + +#### +sub create_gene { + my ($gene_struct_mod, $cds_pointer_lend, $cds_pointer_rend) = @_; + my $strand = $gene_struct_mod->{strand}; + my @exons = @{$gene_struct_mod->{exons}}; + if ($strand eq '-') { + @exons = reverse (@exons); + } + my $mRNA_pointer_lend = 1; + my $mRNA_pointer_rend = 0; + my $gene_obj = new Gene_obj(); + foreach my $coordset_ref (@exons) { + my ($coord1, $coord2) = @$coordset_ref; + my ($end5, $end3) = ($strand eq '+') ? ($coord1, $coord2) : ($coord2, $coord1); + my $exon_obj = new mRNA_exon_obj($end5, $end3); + my $exon_length = ($coord2 - $coord1 + 1); + $mRNA_pointer_rend = $mRNA_pointer_lend + $exon_length - 1; + ## see if cds is within current cDNA range. + if ( $cds_pointer_rend >= $mRNA_pointer_lend && $cds_pointer_lend <= $mRNA_pointer_rend) { #overlap + my $diff = $cds_pointer_lend - $mRNA_pointer_lend; + my $delta_lend = ($diff >0) ? $diff : 0; + $diff = $mRNA_pointer_rend - $cds_pointer_rend; + my $delta_rend = ($diff > 0) ? $diff : 0; + if ($strand eq '+') { + $exon_obj->add_CDS_exon_obj($end5 + $delta_lend, $end3 - $delta_rend); + } else { + $exon_obj->add_CDS_exon_obj($end5 - $delta_lend, $end3 + $delta_rend); + } + } + $gene_obj->add_mRNA_exon_obj($exon_obj); + $mRNA_pointer_lend = $mRNA_pointer_rend + 1; + } + $gene_obj->refine_gene_object(); + #$gene_obj->{strand} = $strand; + print $gene_obj->toString() if $SEE; + return ($gene_obj); +} + + +sub reverse_complement { + my($s) = @_; + my ($rc); + $rc = reverse ($s); + $rc =~tr/ACGTacgtyrkmYRKM/TGCAtgcarymkRYMK/; + return($rc); + } + + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/Fasta_reader.pm b/99.scripts/trinity_utils/PerlLib/Fasta_reader.pm new file mode 100644 index 0000000..28f23b5 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Fasta_reader.pm @@ -0,0 +1,192 @@ +#!/usr/local/bin/perl -w + +# lightweight fasta reader capabilities: +package Fasta_reader; + +use strict; +use warnings; +use Carp; + +sub new { + my ($packagename, $fastaFile) = @_; + + ## note: fastaFile can be a filename or an IO::Handle + + + my $self = { fastaFile => undef,, + fileHandle => undef }; + + bless ($self, $packagename); + + ## create filehandle + my $filehandle = undef; + + if (ref $fastaFile eq 'IO::Handle') { + $filehandle = $fastaFile; + } + else { + if ($fastaFile =~ /\.gz$/) { + open ($filehandle, "gunzip -c $fastaFile | ") or confess "Error, cannot open file $fastaFile using 'gunzip -c'"; + } + else { + open ($filehandle, $fastaFile) or die "Error: Couldn't open $fastaFile\n"; + } + $self->{fastaFile} = $fastaFile; + } + + $self->{fileHandle} = $filehandle; + + return ($self); +} + + + +#### next() fetches next Sequence object. +sub next { + my $self = shift; + my $orig_record_sep = $/; + $/="\n>"; + my $filehandle = $self->{fileHandle}; + my $next_text_input = <$filehandle>; + + if (defined($next_text_input) && $next_text_input !~ /\w/) { + ## must have been some whitespace at start of fasta file, before first entry. + ## try again: + $next_text_input = <$filehandle>; + } + + my $seqobj = undef; + + if ($next_text_input) { + $next_text_input =~ s/^>|>$//g; #remove trailing > char. + $next_text_input =~ tr/\t\n\000-\037\177-\377/\t\n/d; #remove cntrl chars + my ($header, @seqlines) = split (/\n/, $next_text_input); + my $sequence = join ("", @seqlines); + $sequence =~ s/\s//g; + + $seqobj = Sequence->new($header, $sequence); + } + + $/ = $orig_record_sep; #reset the record separator to original setting. + + return ($seqobj); #returns null if not instantiated. +} + + +#### finish() closes the open filehandle to the query database. +sub finish { + my $self = shift; + my $filehandle = $self->{fileHandle}; + close $filehandle; + $self->{fileHandle} = undef; +} + +#### +sub retrieve_all_seqs_hash { + my $self = shift; + + my %acc_to_seq; + + while (my $seq_obj = $self->next()) { + my $acc = $seq_obj->get_accession(); + my $sequence = $seq_obj->get_sequence(); + + $acc_to_seq{$acc} = $sequence; + } + + return(%acc_to_seq); +} + + + +############################################## +package Sequence; +use strict; + +sub new { + my ($packagename, $header, $sequence) = @_; + + ## extract an accession from the header: + my ($acc, $rest) = split (/\s+/, $header, 2); + + my $self = { accession => $acc, + header => $header, + sequence => $sequence, + filename => undef }; + bless ($self, $packagename); + return ($self); +} + +#### +sub get_accession { + my $self = shift; + return ($self->{accession}); +} + +#### +sub get_header { + my $self = shift; + return ($self->{header}); +} + +#### +sub get_sequence { + my $self = shift; + return ($self->{sequence}); +} + +#### +sub get_FASTA_format { + my $self = shift; + my %settings = @_; + + my $fasta_line_len = $settings{fasta_line_len} || 60; + + my $header = $self->get_header(); + my $sequence = $self->get_sequence(); + if ($fasta_line_len > 0) { + $sequence =~ s/(\S{$fasta_line_len})/$1\n/g; + chomp $sequence; + } + my $fasta_entry = ">$header\n$sequence\n"; + return ($fasta_entry); +} + + +#### +sub write_fasta_file { + my $self = shift; + my $filename = shift; + + my ($accession, $header, $sequence) = ($self->{accession}, $self->{header}, $self->{sequence}); + + my $fasta_entry = $self->get_FASTA_format(); + + my $tempfile; + if ($filename) { + $tempfile = $filename; + } else { + my $acc = $accession; + $acc =~ s/\W/_/g; + $tempfile = "$acc.fasta"; + } + + open (TMP, ">$tempfile") or die "ERROR! Couldn't write a temporary file in current directory.\n"; + print TMP $fasta_entry; + close TMP; + return ($tempfile); +} + +#### +sub get_core_read_name { + my $self = shift; + + my $acc = $self->get_accession(); + $acc =~ s|/[12]$||; + return($acc); +} + + +1; #EOM + + diff --git a/99.scripts/trinity_utils/PerlLib/Fasta_retriever.pm b/99.scripts/trinity_utils/PerlLib/Fasta_retriever.pm new file mode 100644 index 0000000..771475b --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Fasta_retriever.pm @@ -0,0 +1,126 @@ +package Fasta_retriever; + +use strict; +use warnings; +use Carp; +use threads; +use threads::shared; + + +my $LOCKVAR :shared; +our $DEBUG = 0; + +sub new { + my ($packagename) = shift; + my $filename = shift; + + unless ($filename) { + confess "Error, need filename as param"; + } + + my $self = { filename => $filename, + acc_to_pos_index => undef, + fh => undef, + }; + + my %acc_to_pos_index :shared; + + $self->{acc_to_pos_index} = \%acc_to_pos_index; + + bless ($self, $packagename); + + $self->_init(); + + + return($self); +} + + +sub _init { + my $self = shift; + + my $filename = $self->{filename}; + + # use a samtools faidx index if available + my $index_file = "$filename.fai"; + if (-s $index_file) { + open(my $fh, $index_file) or die "Error, cannot open file: $index_file"; + while(<$fh>) { + chomp; + my @x = split(/\t/); + my $acc = $x[0]; + my $file_pos = $x[2]; + $self->{acc_to_pos_index}->{$acc} = $file_pos; + } + close $fh; + } + else { + print STDERR "-missing faidx file: $index_file, extracting positions directly.\n"; + print STDERR "-Fasta_retriever:: begin initializing for $filename\n"; + + open (my $fh, $filename) or die $!; + $self->{fh} = $fh; + while (<$fh>) { + if (/>(\S+)/) { + my $acc = $1; + my $file_pos = tell($fh); + $self->{acc_to_pos_index}->{$acc} = $file_pos; + } + } + print STDERR "-Fasta_retriever:: done initializing for $filename\n"; + } + + return; +} + +sub refresh_fh { + my $self = shift; + + open (my $fh, $self->{filename}) or die "Error, cannot open file : " . $self->{filename}; + $self->{fh} = $fh; + + return $fh; +} + + +sub get_seq { + my $self = shift; + my $acc = shift; + + unless (defined $acc) { + confess "Error, need acc as param"; + } + + + my $seq = ""; + + { + lock $LOCKVAR; + + my $file_pos = $self->{acc_to_pos_index}->{$acc} or confess "Error, no seek pos for acc: $acc"; + + my $fh = $self->refresh_fh(); + seek($fh, $file_pos, 0); + + print STDERR "seeking $acc -> $file_pos\n" if $DEBUG; + + while (<$fh>) { + if (/^>/) { + print STDERR " reached $_, stopping\n" if $DEBUG; + last; + } + my $seq_part = $_; + $seq_part =~ s/\s//g; + $seq .= $seq_part; + } + print STDERR "-done seeking $acc\n\n" if $DEBUG; + + } + + return($seq); +} + + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/Fastq_reader.pm b/99.scripts/trinity_utils/PerlLib/Fastq_reader.pm new file mode 100644 index 0000000..5f36dbe --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Fastq_reader.pm @@ -0,0 +1,173 @@ +package Fastq_reader; + +use strict; +use warnings; +use Carp; + +sub new { + my ($packagename, $fastqFile) = @_; + + ## note: fastqFile can be a filename or an IO::Handle + + + my $self = { fastqFile => undef, + fileHandle => undef }; + + bless ($self, $packagename); + + ## create filehandle + my $filehandle = undef; + + if (ref $fastqFile eq 'IO::Handle') { + $filehandle = $fastqFile; + } + else { + if ( $fastqFile =~ /\.gz$/ ) { + ## TODO: need to handle the failure case, since not picked up here as I thought it would!! + open ($filehandle, "gunzip -c $fastqFile | ") or die "Error: Couldn't open compressed $fastqFile\n"; + } + elsif ($fastqFile =~ /\.bz2$/) { + open ($filehandle, "bunzip2 -c $fastqFile | ") or die "Error, couldn't open compressed $fastqFile $!"; + + } else { + open ($filehandle, $fastqFile) or die "Error: Couldn't open $fastqFile\n"; + } + + $self->{fastqFile} = $fastqFile; + } + + $self->{fileHandle} = $filehandle; + + return ($self); +} + + + +#### next() fetches next Sequence object. +sub next { + my $self = shift; + + my $filehandle = $self->{fileHandle}; + my $next_text_input = ""; + + if (! eof($filehandle)) { + for (1..4) { + $next_text_input .= <$filehandle>; + } + } + + my $read_obj = undef; + + if ($next_text_input) { + + eval { + $read_obj = Fastq_record->new($next_text_input); + }; + if ($@) { + confess "Error, $@ :: fastq file: " . $self->{fastqFile}; + } + } + + return ($read_obj); #returns null if not instantiated. + +} + + +#### finish() closes the open filehandle to the query database. +sub finish { + my $self = shift; + my $filehandle = $self->{fileHandle}; + close $filehandle; + $self->{fileHandle} = undef; +} + + +############################################## +package Fastq_record; + +use strict; +use warnings; +use Carp; + +sub new { + my $packagename = shift; + my ($text_lines) = @_; + + my @split_text = split(/\n/, $text_lines); + unless (scalar @split_text == 4) { + confess "Error, fastQ entry doesn't have 4 lines: " . $text_lines; + } + + my ($name_line, $seq_line, $plus, $qual_line) = @split_text; + + unless ($name_line =~ /^\@/) { + confess "Error, cannot identify first line as read name line: " . $text_lines; + } + + my ($read_name, $rest) = split(/\s+/, $name_line); + $read_name =~ s/^\@//; + + + my $pair_dir = 0; # assume single + + if ($read_name =~ /^(\S+)\/([12])$/) { + $read_name = $1; + $pair_dir = $2; + } + elsif (defined($rest) && $rest =~ /^([12]):/) { + $pair_dir = $1; + } + + + my $self = { core_read_name => $read_name, + pair_dir => $pair_dir, # (0, 1, or 2), with 0 = unpaired. + sequence => $seq_line, + quals => $qual_line, + record => $text_lines, + }; + + + bless ($self, $packagename); + return ($self); +} + +#### +sub get_core_read_name { + my $self = shift; + return ($self->{core_read_name}); +} + +#### +sub get_full_read_name { + my $self = shift; + + my $read_name = $self->{core_read_name}; + if ($self->{pair_dir}) { + return(join("/", $read_name, $self->{pair_dir})); + } +} + +#### +sub get_sequence { + my $self = shift; + return($self->{sequence}); +} + +#### +sub get_quals { + my $self = shift; + return($self->{quals}); +} + +#### +sub get_fastq_record { + my $self = shift; + return($self->{record}); +} + + + + +1; #EOM + + diff --git a/99.scripts/trinity_utils/PerlLib/GFF3_alignment_utils.pm b/99.scripts/trinity_utils/PerlLib/GFF3_alignment_utils.pm new file mode 100644 index 0000000..35f6755 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/GFF3_alignment_utils.pm @@ -0,0 +1,186 @@ +#!/usr/bin/env perl + +package GFF3_alignment_utils; + +use strict; +use warnings; +use Carp; + +use Gene_obj; +use Gene_obj_indexer; +use CDNA::Alignment_segment; +use CDNA::CDNA_alignment; +use File::Basename; + +__run_test() unless caller; + +sub index_alignment_objs { + my ($gff3_alignment_file, $genome_alignment_indexer_href) = @_; + + unless ($gff3_alignment_file && -s $gff3_alignment_file) { + confess "Error, cannot find or open file $gff3_alignment_file"; + } + unless (ref $genome_alignment_indexer_href) { + confess "Error, need genome indexer href as param "; + } + + + my %genome_trans_to_alignment_segments; + my %trans_to_gene_id; + + + open (my $fh, $gff3_alignment_file) or die "Error, cannot open file $gff3_alignment_file"; + while (<$fh>) { + chomp; + + unless (/\w/) { next; } + + my @x = split(/\t/); + + unless (scalar (@x) >= 8 && $x[8] =~ /ID=/) { + print STDERR "ignoring line: $_\n"; + next; + } + + my $scaff = $x[0]; + my $type = $x[2]; + my $lend = $x[3]; + my $rend = $x[4]; + my $per_id = $x[5]; + if ($per_id eq ".") { $per_id = 100; } # making an assumption here. + + my $orient = $x[6]; + + my $info = $x[8]; + + my @parts = split(/;/, $info); + my %atts; + foreach my $part (@parts) { + $part =~ s/^\s+|\s+$//; + $part =~ s/\"//g; + my ($att, $val) = split(/=/, $part); + + if (exists $atts{$att}) { + die "Error, already defined attribute $att in $_"; + } + + $atts{$att} = $val; + } + + my $gene_id = $atts{ID} or die "Error, no gene_id at $_"; + my $trans_id = $atts{Target} or die "Error, no trans_id at $_"; + { + my @pieces = split(/\s+/, $trans_id); + $trans_id = shift @pieces; + } + + my ($end5, $end3) = ($orient eq '+') ? ($lend, $rend) : ($rend, $lend); + + $info =~ /Target=\S+ (\d+) (\d+)/ or die "Error, cannot extract match coordinates from info: $info"; + my $cdna_seg_lend = $1; + my $cdna_seg_rend = $2; + + ($cdna_seg_lend, $cdna_seg_rend) = sort {$a<=>$b} ($cdna_seg_lend, $cdna_seg_rend); # always + orient for transcript coords. + + + my $alignment_segment = new CDNA::Alignment_segment($end5, $end3, $cdna_seg_lend, $cdna_seg_rend, $per_id); + + + push (@{$genome_trans_to_alignment_segments{$scaff}->{$trans_id}}, $alignment_segment); + + $trans_to_gene_id{$trans_id} = $gene_id; + + } + + + my %scaff_to_align_list; + + + ## Output genes in gff3 format: + + foreach my $scaff (sort keys %genome_trans_to_alignment_segments) { + + my @alignment_accs = keys %{$genome_trans_to_alignment_segments{$scaff}}; + + foreach my $alignment_acc (@alignment_accs) { + + my $segments_aref = $genome_trans_to_alignment_segments{$scaff}->{$alignment_acc}; + + ## determine cdna length + my @cdna_coords; + foreach my $segment (@$segments_aref) { + push (@cdna_coords, $segment->get_mcoords()); + } + @cdna_coords = sort {$a<=>$b} @cdna_coords; + my $max_coord = pop @cdna_coords; + + my $cdna_alignment_obj = new CDNA::CDNA_alignment($max_coord, $segments_aref); + $cdna_alignment_obj->set_acc($alignment_acc); + $cdna_alignment_obj->{genome_acc} = $scaff; + + my $gene_id = $trans_to_gene_id{$alignment_acc} or confess "Error no gene_id for acc: $alignment_acc"; + + $cdna_alignment_obj->{gene_id} = $gene_id; + + + $cdna_alignment_obj->{source} = basename($gff3_alignment_file); + + if (ref $genome_alignment_indexer_href eq "Gene_obj_indexer") { + $genome_alignment_indexer_href->store_gene($alignment_acc, $cdna_alignment_obj); + + } + else { + + $genome_alignment_indexer_href->{$alignment_acc} = $cdna_alignment_obj; + } + push (@{$scaff_to_align_list{$scaff}}, $alignment_acc); + } + } + + return(%scaff_to_align_list); +} + + + +################# +## Testing +################# + + +sub __run_test { + + my $usage = "usage: $0 file.alignment.gff3\n\n"; + + my $gff3_file = $ARGV[0] or die $usage; + + my $indexer = {}; + my %scaff_to_alignments = &index_alignment_objs($gff3_file, $indexer); + + + foreach my $scaffold (keys %scaff_to_alignments) { + + my @align_ids = @{$scaff_to_alignments{$scaffold}}; + + foreach my $align_id (@align_ids) { + my $cdna_obj = $indexer->{$align_id}; + + print $cdna_obj->toString(); + } + } + + + + exit(0); + + + +} + + + + + + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/GFF3_utils.pm b/99.scripts/trinity_utils/PerlLib/GFF3_utils.pm new file mode 100644 index 0000000..aa77d12 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/GFF3_utils.pm @@ -0,0 +1,262 @@ +#!/usr/local/bin/perl + +package main; +our $SEE; + + +package GFF3_utils; + +use strict; +use warnings; +use Gene_obj; +use Gene_obj_indexer; +use Carp; +use URI::Escape; +use Data::Dumper; + + +#### +sub index_GFF3_gene_objs { + + my ($gff_filename, $gene_obj_indexer, $contig_id) = @_; + # contig_id is optional. + + + my $hash_mode = 0; + if (ref $gene_obj_indexer eq 'HASH') { + $hash_mode = 1; + } + + ## note can use either a gene_obj_indexer or a hash reference. + + my %gene_coords; + my %asmbl_id_to_gene_id_list; + my %transcript_to_gene; + my %cds_phases; + + my %gene_names; + my %loci; + + open (my $fh, $gff_filename) or die $!; + + my %gene_id_to_source_type; + + my %source_tracker; + + my $counter = 0; + # print STDERR "\n-parsing file $gff_filename\n"; + while (<$fh>) { + + chomp; + + unless (/\w/) { next;} # empty line + + if (/^\#/) { next; } # comment entry in gff3 + + my @x = split (/\t/); + + unless (scalar @x >= 9) { + print STDERR "-ignoring line $_\n"; + next; + } + + my ($asmbl_id, $source, $feat_type, $lend, $rend, $orient, $cds_phase, $gene_info) = ($x[0], $x[1], $x[2], $x[3], $x[4], $x[6], $x[7], $x[8]); + + if ($contig_id && $asmbl_id ne $contig_id) { next; } + + unless ($feat_type) { die "Error, $_, no feat_type: line\[$_\]"; } + + unless ($feat_type =~ /^(gene|mRNA|CDS|exon)$/) { next;} + + $gene_info = uri_unescape($gene_info); + + $gene_info =~ /ID=\"?([^;\s\"]+)\"?;?/; + my $id = $1 or die "Error, couldn't get the id field $_"; + + if (exists $source_tracker{$id} && $source_tracker{$id} ne $source) { + confess "Error, gene ID $id is given source $source when previously encountered with source $source_tracker{$id} "; + } + + if ($feat_type eq 'gene') { + my $gene_name = ""; + if ($gene_info =~ /Name=\"?([^\;\"]+)\"?/) { + $gene_name = $1; + } + else { + $gene_name = ""; + } + + if ($gene_info =~ /Note=\"?([^\;\"]+)\"?/) { + $gene_name .= " $1"; + } + + $gene_names{$id} = $gene_name; + + } + + if ($gene_info =~ /Alias=([^;]+)/) { + my $locus = $1; + $loci{$id} = $locus; + } + + + if ($feat_type eq 'gene') { next;} ## beyond this pt, gene is not needed. + + $gene_info =~ /Parent=\"?([^;\s\"]+)\"?;?/; + my $parent = $1 or die "Error, couldn't get the parent info $_"; + + # print "id: $id, parent: $parent\n"; + + if ($feat_type eq 'mRNA') { + ## just get the identifier info + $transcript_to_gene{$id} = $parent; + next; + } + + my $transcript_id_listing = $parent; + + my @transcript_ids = split(/,/, $transcript_id_listing); + foreach my $transcript_id (@transcript_ids) { + + my $gene_id = $transcript_to_gene{$transcript_id}; + unless (defined $gene_id) { + print STDERR "Error, no gene feature found for $transcript_id.... ignoring feature.\n"; + next; + } + + + $gene_id_to_source_type{$gene_id} = $source; + + my ($end5, $end3) = ($orient eq '+') ? ($lend, $rend) : ($rend, $lend); + + $gene_coords{$asmbl_id}->{$gene_id}->{$transcript_id}->{$feat_type}->{$end5} = $end3; + # print "$asmbl_id, $gene_id, $transcript_id, $feat_type, $end5, $end3\n"; + + if ($cds_phase =~ /^\d+$/) { + $cds_phases{$gene_id}->{$transcript_id}->{$end5} = int($cds_phase); + } + } + + } + close $fh; + + ## + # print STDERR "\n-caching genes.\n"; + foreach my $asmbl_id (sort keys %gene_coords) { + my $genes_href = $gene_coords{$asmbl_id}; + + foreach my $gene_id (keys %$genes_href) { + print STDERR "\r-indexing [$gene_id] " if $SEE; + my $transcripts_href = $genes_href->{$gene_id}; + + my @gene_objs; + + foreach my $transcript_id (keys %$transcripts_href) { + + my $cds_coords_href = $transcripts_href->{$transcript_id}->{CDS} || {}; # could be a noncoding transcript w/ no CDS + my $exon_coords_href = $transcripts_href->{$transcript_id}->{exon}; + + unless (ref $exon_coords_href) { + print STDERR Dumper ($transcripts_href); + die "Error, missing exon coords for $transcript_id, $gene_id\n"; + } + + my $gene_obj = new Gene_obj(); + + + if (scalar (keys %$cds_coords_href) == 1) { + + ## could be that only the cds span was provided. + ## break it up across the exon segments + + my ($cds_lend, $cds_rend) = sort {$a<=>$b} %$cds_coords_href; + my @exon_coords; + my $orient; + foreach my $exon_end5 (keys %$exon_coords_href) { + my $exon_end3 = $exon_coords_href->{$exon_end5}; + push (@exon_coords, [$exon_end5, $exon_end3]); + if ($exon_end5 < $exon_end3) { + $orient = '+'; + } + elsif ($exon_end5 > $exon_end3) { + $orient = '-'; + } + } + + $gene_obj->build_gene_obj_exons_n_cds_range(\@exon_coords, $cds_lend, $cds_rend, $orient); + } + else { + + ## cds and exons specified separately + + $gene_obj->populate_gene_obj($cds_coords_href, $exon_coords_href); + } + + $gene_obj->{Model_feat_name} = $transcript_id; + $gene_obj->{TU_feat_name} = $gene_id; + $gene_obj->{asmbl_id} = $asmbl_id; + + if (my $gene_locus = $loci{$gene_id}) { + $gene_obj->{pub_locus} = $gene_locus; + } + if (my $transcript_locus = $loci{$transcript_id}) { + $gene_obj->{model_pub_locus} = $transcript_locus; + } + + + $gene_obj->{com_name} = $gene_names{$gene_id} || $transcript_id; + + $gene_obj->{source} = $gene_id_to_source_type{$gene_id}; + + ## set CDS phase info if available from the gff + my $cds_phases_href = $cds_phases{$gene_id}->{$transcript_id}; + if (ref $cds_phases_href) { + ## set the cds phases + my @exons = $gene_obj->get_exons(); + foreach my $exon (@exons) { + if (my $cds = $exon->get_CDS_obj()) { + my ($end5, $end3) = $cds->get_coords(); + my $phase = int($cds_phases_href->{$end5}); + unless ($phase == 0 || $phase == 1 || $phase == 2) { + confess "Error, should have phase set for cds $gene_id $transcript_id $end5, but I do not know what $phase is. "; + } + $cds->set_phase($phase); + } + } + } + + push (@gene_objs, $gene_obj); + } + + ## want single gene that includes all alt splice variants here + my $template_gene_obj = shift @gene_objs; + foreach my $other_gene_obj (@gene_objs) { + $template_gene_obj->add_isoform($other_gene_obj); + } + + $template_gene_obj->refine_gene_object(); + + if ($hash_mode) { + $gene_obj_indexer->{$gene_id} = $template_gene_obj; + } + else { + $gene_obj_indexer->store_gene($gene_id, $template_gene_obj); + } + + print "GFF3_utils: stored $gene_id\n" if $SEE; + + # add to gene list for asmbl_id + my $gene_list_aref = $asmbl_id_to_gene_id_list{$asmbl_id}; + unless (ref $gene_list_aref) { + $gene_list_aref = $asmbl_id_to_gene_id_list{$asmbl_id} = []; + } + push (@$gene_list_aref, $gene_id); + } + } + print STDERR "\n"; + return (\%asmbl_id_to_gene_id_list); +} + + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/GFF_maker.pm b/99.scripts/trinity_utils/PerlLib/GFF_maker.pm new file mode 100644 index 0000000..aae4e2a --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/GFF_maker.pm @@ -0,0 +1,88 @@ +#!/usr/local/bin/perl + +package main; +our $SEE; + +package GFF_maker; + +use strict; +use warnings; +use Carp; +use Data::Dumper; + +#### +sub get_GFF_line { + my $input_href = shift; + + ## field 1: Sequence Identifier + my $seq_id = $input_href->{seq_id} or confess "need seq_id " . Dumper($input_href); + + ## field 2: Source (default '.') + my $source = $input_href->{source} || '.'; + + ## field 3: type (use SO:id syntax or corresponding token) + ## -could do some better validation here. + my $type = $input_href->{type}; + + ## field 4: starting coordinate (lend) + my $lend = $input_href->{lend}; + + ## field 5: ending coordinate (rend) + my $rend = $input_href->{rend}; + + unless (defined $lend && defined $rend) { + confess "lend or rend coordinate not defined: " . Dumper($input_href); + } + + if ($lend == 0) { + print STDERR "Warning, lend coordinate is zero, changing it to base 1.\n"; + print STDERR Dumper($input_href); + $lend = 1; + } + + unless ($lend <= $rend) { + confess " lend > rend, not allowed. " . Dumper($input_href); + } + + ## field 6: score + my $score = $input_href->{score} || '.'; + + ## field 7: strand + my $strand = $input_href->{strand}; + unless ($strand eq '+' || $strand eq '-') { + confess "need strand " . Dumper ($input_href); + } + + ## field 8: strand + my $phase = $input_href->{phase}; + unless (defined $phase) { + $phase = '.'; + } + + ## field 9: attributes + my $attributes = $input_href->{attributes} or confess "need attributes " . Dumper($input_href); + unless ($attributes =~ /ID=/) { + confess "attributes requires ID field. " . Dumper($input_href); + } + + my $textline = sprintf("%s\t%s\t%s\t%i\t%i\t%s\t%s\t%s\t%s\n", + $seq_id, + $source, + $type, + $lend, + $rend, + $score, + $strand, + $phase, + $attributes); + + return ($textline); + +} + + + + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/GTF.pm b/99.scripts/trinity_utils/PerlLib/GTF.pm new file mode 100644 index 0000000..b5f2505 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/GTF.pm @@ -0,0 +1,4280 @@ +# -*- Mode: Perl; tab-width: 8; perl-indent-level: 2; indent-tabs-mode: nil -*- +use strict; +############################################################################### +# GTF +############################################################################### +# The GTF object parses and stores all data from a gtf file. +package GTF; +use Carp; +# GTF::new(hash) +# This is the contructor for GTF objects. It takes one required argument +# which is a hash containing the following optional fields: +# gtf_filename: The filename of the gtf file to be loaded. If no filename +# is given it just creates an empty object. +# seq_filename: The filename of the sequence corresponding to this gtf file. +# If given additional validity checks are made, additional stats +# are kept and the transcript can be output. +# tx_out_fh: The filehandle to output the genes transcripts to. Only works +# If seq_filename is given. +# conseq_filename: The filename of the conservation sequence corresponding to +# this gtf file. If given additional stats are kept. +# warning_fh: The filehandle to output gtf format warnings to. If none +# is given warnings are disregarded. +sub new { + my ($info) = @_; + my $gtf = bless {Genes => [], Transcripts => [], Exons => [], CDS => [], + Inter => [], Inter_CNS => [], Filename => "", Sequence => "", + Comments => "", Modified => 0}; + $gtf->{Total_Conseq} = {0 => -1, + 1 => -1, + 2 => -1}; + $gtf->{Total_Seq} = {A => -1, + C => -1, + G => -1, + T => -1, + N => -1}; + if(defined($info->{seq_filename})){ + if(-e $info->{seq_filename}){ + $gtf->{Sequence} = $info->{seq_filename}; + } + else{ + $gtf->{Sequence} = 0; + print STDERR "GTF: $info->{seq_filename} does not exist.\n"; + } + } + else{ + $gtf->{Sequence} = 0; + } + if(defined($info->{strict})){ + $gtf->{Strict} = $info->{strict}; + } + else{ + $gtf->{Strict} = 0; + } + if(defined($info->{tx_out_fh})){ + if(defined($gtf->{Sequence})){ + $gtf->{Tx} = $info->{tx_out_fh}; + } + else{ + $gtf->{Tx} = 0; + print STDERR "GTF: Cannot ouput transcripts without the sequence.\n"; + } + } + else{ + $gtf->{Tx} = 0; + } + if(defined($info->{conseq_filename})){ + if(-e $info->{conseq_filename}){ + $gtf->{Conseq} = $info->{conseq_filename}; + } + else{ + $info->{Conseq} = 0; + print STDERR "GTF: $info->{conseq_filename} does not exist.\n"; + } + } + else{ + $info->{Conseq} = 0; + } + if((defined($info->{inframe_stops})) && ($info->{inframe_stops})){ + $gtf->{Inframe_Stops} = 1; + } + else{ + $gtf->{Inframe_Stops} = 0; + } + if((defined($info->{fix_gtf})) && + ($info->{fix_gtf} == 1)){ + $gtf->{Fix_GTF} = 1; + } + else{ + $gtf->{Fix_GTF} = 0; + } + if(defined($info->{warning_skips})){ + $gtf->{Warning_Skips} = $info->{warning_skips}; + } + else{ + $gtf->{Warning_Skips} = []; + } + if(defined($info->{bad_list})){ + $gtf->{Bad_List} = $info->{bad_list}; + } + else{ + $gtf->{Bad_List} = []; + } + if(defined($info->{detailed_error_count})){ + $gtf->{Detailed_Error_Count} = $info->{detailed_error_count}; + } + if(defined($info->{mark_ase})) { + $gtf->{Mark_ASE} = $info->{mark_ase}; + } + else { + $gtf->{Mark_ASE} = 0; + } + if(defined($info->{gtf_filename})){ + if(-e $info->{gtf_filename}){ + $gtf->{Filename} = $info->{gtf_filename}; + if((defined($info->{no_check})) && ($info->{no_check})){ + $gtf->_load_no_check($info->{gtf_filename}); + } + else{ + $gtf->_parse_file($info->{warning_fh}); + } + if($gtf->{Mark_ASE}) { + $gtf->mark_ase; + } + } + else{ + print STDERR "GTF: $info->{gtf_filename} does not exist.\n"; + } + } + else{ + $gtf->{Filename} = ""; + } + return $gtf; +} + +# GTF::_update +# This function resorts all lists. It is called when a list is requested. +# It does nothing unless the $gtf->{Modified} value is 1. Sets +# $gene->{Modified} to 0; +sub _update{ + my ($gtf) = @_; + unless($gtf->{Modified}){ + return; + } + $gtf->{Modified} = 0; + my $genes = $gtf->{Genes}; + my $txs = $gtf->{Transcripts}; + my $cds = $gtf->{CDS}; + my $inter = $gtf->{Inter}; + $genes = [sort {$a->start <=> $b->start} @$genes]; + $txs = [sort {$a->start <=> $b->start} @$txs]; + $cds = [sort {$a->start <=> $b->start} @$cds]; + $inter = [sort {$a->start <=> $b->start} @$inter]; + $gtf->{Genes} = $genes; + $gtf->{Transcripts} = $txs; + $gtf->{CDS} = $cds; + $gtf->{Inter} = $inter; +} + +# GTF::genes() +# This function returns an array of refernces to GTF::Gene objects. +# The array contains one reference for each gene in the gtf file. +# The array is sorted by the start position of each gene. +sub genes{ + my ($gtf) = @_; + $gtf->_update; + return $gtf->{Genes}; +} + +sub transcripts{ + my ($gtf) = @_; + $gtf->_update; + return $gtf->{Transcripts}; +} + +# GTF::add_gene(gene_ref) +# This function take a reference to a GTF::Gene object and adds it to the +# list of genes stored by this object. +sub add_gene{ + my ($gtf,$gene) = @_; + my $genes = $gtf->{Genes}; + my $txs = $gtf->{Transcripts}; + my $cds = $gtf->{CDS}; + push @$genes, $gene; + push @$txs, @{$gene->transcripts}; + push @$cds, @{$gene->cds}; + $gtf->{Modified} = 1; +} + +# GTF::add_feature(feature_ref,seqname,source,strand,id) +# This function will take a reference to an intergenic feature (such as +# inter or inter_CNS) and add it to the appropriate feature list. +# Pass a sequence name, the strand ("+" is OK for intergenic) +# and an ID to identify the feature. +# Note that the feature must be of the types inter or inter_CNS. +sub add_feature{ + my ($gtf,$feature,$seqname,$source,$strand,$id) = @_; + die "Only inter or inter_CNS features may be directly added to a GTF.\n" + if($feature->type ne "inter" && $feature->type ne "inter_CNS"); + my $gene = GTF::Gene::new($id,$seqname,$source,$strand); + my $tx = GTF::Transcript::new($id); + $gene->add_transcript($tx); + $tx->add_feature($feature); + my $features; + if($feature->type eq "inter") { $features = $gtf->{Inter} } + elsif($feature->type eq "inter_CNS") { $features = $gtf->{Inter_CNS} } + push @$features, $feature; + $gtf->{Modified} = 1; +} + +# GTF::set_genes(gene_list) +# This function take a reference to a list of GTF::Gene objects and adds each to the +# list of genes stored by this object. +sub set_genes{ + my ($gtf,$genes) = @_; + $gtf->{Genes} = $genes; + my @txs; + my @cds; + foreach my $gene (@$genes){ + push @txs, @{$gene->transcripts}; + push @cds, @{$gene->cds}; + } + $gtf->{Transcripts} = \@txs; + $gtf->{CDS} = \@cds; + $gtf->{Modified} = 1; +} + +# GTF::remove_gene(gene_id) +# This function takes a gene_id and removes the gene with that gene_id +# from the list of genes stored by this object, if a gene with that gene_id +# can be found. Otherwise does nothing. Returns true if a gene was removed +# and false if no gene with that gene_id was found. +sub remove_gene{ + my ($gtf,$gene_id) = @_; + my $genes = $gtf->{Genes}; + my $removed = 0; + my @new_genes; + my @new_txs; + my @new_cds; + foreach my $gene (@$genes){ + if($gene->gene_id eq $gene_id){ + $removed++; + } + else{ + push @new_genes, $gene; + push @new_txs, @{$gene->transcripts}; + push @new_cds, @{$gene->cds}; + } + } + if($removed){ + $gtf->{Genes} = \@new_genes; + $gtf->{Transcripts} = \@new_txs; + $gtf->{CDS} = \@new_cds; + } + return $removed; +} + +# GTF::remove_feature(feature) +# This function will remove a feature from the list of features. +# Note the feature must be of type inter or inter_CNS. +sub remove_feature { + my($gtf,$feature) = @_; + my $removed = 0; + my $feature_key; + if($feature->type eq "inter") { + $feature_key = "Inter"; + } + elsif($feature->type eq "inter_CNS") { + $feature_key = "Inter_CNS"; + } + my @new_features; + foreach my $stored_feature (@{$gtf->{$feature_key}}) { + if($stored_feature == $feature) { + $removed++; + } + else { + push @new_features, $stored_feature; + } + } + if($removed) { + $gtf->{$feature_key} = \@new_features; + } + return $removed; +} + +# GTF::set_filename(filename) +# Sets the filename to be returned by the filename function +sub set_filename{ + my ($gtf,$filename) = @_; + $gtf->{Filename} = $filename; +} + +# GTF::offset(ammount) +# This function will offset all genes in this gene by the given ammount +sub offset{ + my ($gtf,$offset) = @_; + unless(defined($offset)){ + print STDERR "Undefined value passed to GTF::offset.\n"; + return; + } + my $genes = $gtf->{Genes}; + foreach my $gene (@$genes){ + $gene->offset($offset); + } + $gtf->{Modified} = 1; +} + +# GTF::reverse_complement(seq_length) +# This function takes the length of the sequence this gtf file was based on and +# and reverse complements everything in the file. So the positive strand becomes +# the negative strand and the files starts at seq_length and moves downward to 0. +sub reverse_complement{ + my ($gtf, $seq_length) = @_; + unless(defined($seq_length)){ + print STDERR + "Undefined value for seq_length passed to GTF::reverse_complement.\n"; + } + my $genes = $gtf->{Genes}; + foreach my $gene(@$genes){ + $gene->reverse_complement($seq_length); + } + $gtf->{Modified} = 1; +} + +# GTF::cds() +# This function returns an array of refernces to GTF::Feature objects. +# The array contains one refernece to each CDS feature in the gtf file. +# The array is sorted by the of each CDS. +sub cds{ + my ($gtf) = @_; + $gtf->_update; + return $gtf->{CDS}; +} + +sub inter { + my ($gtf) = @_; + $gtf->_update; + return $gtf->{Inter}; +} + +sub inter_cns { + my ($gtf) = @_; + $gtf->_update; + return $gtf->{Inter_CNS}; +} + +# GTF::mark_ase() +# If you are looking for alternative splicing events in your GTF file, +# call this function to marke them. This function is separate from the +# _update methods in order to keep the running time low. + +sub mark_ase { + my ($gtf) = @_; + + #set the alternative splicing event flags for each feature + #currently, only optional coding exons + + # The main idea is to check all pairs of transcripts in a gene, + # and find any incident where there are two exons that have the same + # end coord and two exons which have the same start coord + # and that an exon in only one of the transcripts lies between them. + # It is accomplished by sorting all the exons in both transcripts + # and checking for a supercasette, i.e. + # + # ===|||||||||===||||||||======||||||||||==== + # ==||||||||||=================||||||||||||== + # <---------------> + # supercassette + # + # Note that it doesn't matter if the opposite ends of the exon are equal. + + # rpz + + my @genes = @{$gtf->genes}; + for my $gene(@{$gtf->genes}) { + my @txs = (@{$gene->transcripts}); + if(scalar(@txs) > 1) { + for (my $i = 0; $i < scalar(@txs)-1; $i++) { + next if(scalar(@{$txs[$i]->cds}) == 0); + for (my $j = $i+1; $j < scalar(@txs); $j++) { + next if(scalar(@{$txs[$j]->cds}) == 0); + + my @ocds = sort {$a->start <=> $b->start || $a->stop <=> $b->stop} + (@{$txs[$i]->cds}, @{$txs[$j]->cds}); + + for(my $k = 4; $k < scalar(@ocds); $k++) { + if( $ocds[$k-4]->stop == $ocds[$k-3]->stop && + $ocds[$k-1]->start == $ocds[$k]->start) { + if($ocds[$k-2]->length%3 == 0) { + $ocds[$k-2]->set_ase("InframeOptional"); + } + else { + $ocds[$k-2]->set_ase("FrameshiftOptional"); + } + } + } + } + } + } + } + +} + +# GTF::infer_exons() +# Skips infering if Exons already exist +# Output is undefined if features overlap, except for start_codon +# Currently knows about utr5, start,stop,cds, utr3 +sub infer_exons { + my ($gtf) = @_; + $gtf->_update; + + for my $tx (@{$gtf->transcripts}) { + next if (@{$tx->exons}); + + if ($tx->strand eq '+') { + for ( @{$tx->utr5}, @{$tx->cds}, @{$tx->stop_codons}, @{$tx->utr3} ) { + + if (!@{$tx->exons} || ${$tx->exons}[-1]->stop + 1 < $_->start ) { + $tx->add_feature( GTF::Feature::new('exon', $_->start, $_->stop, 0, 0) ); + + } elsif ( ${$tx->exons}[-1]->stop + 1 eq $_->start ) { + ${$tx->exons}[-1]->set_stop($_->stop); + } + } + } else { + for ( @{$tx->utr3}, @{$tx->stop_codons}, @{$tx->cds}, @{$tx->utr5} ) { + + if (!@{$tx->exons} || ${$tx->exons}[-1]->stop + 1 < $_->start ) { + $tx->add_feature(GTF::Feature::new('exon', $_->start, $_->stop, 0, 0)); + } elsif ( ${$tx->exons}[-1]->stop + 1 eq $_->start ) { + ${$tx->exons}[-1]->set_stop($_->stop); + } + } + } + } +} + +# GTF::remove_exons() +# remove exons from all transcripts +sub remove_exons { + my ($gtf) = @_; + $gtf->_update; + + $_->{Exons} = [] for (@{$gtf->transcripts}); +} + +# GTF::filename() +# This function returns the filename of the gtf file. +sub filename {shift->{Filename}} + +# GTF::comments() +# This function returns all full line commentsfrom the gtf file. +sub comments {shift->{Comments}}; + +# GTF::conservation_count() +# Returns a hash containing counts for each conservation character (0,1,2) +sub conservation_count {shift->{Total_Conseq}}; + +# GTF::sequence_count() +# Returns a hash containing counts for each sequence character (A,C,T,G,N,X), +# Where X is all other characters +sub sequence_count {shift->{Total_Seq}}; + +# GTF::output_gtf_file([file_handle]); +# This function moves through each gene in the gtf file and outputs +# all data in valid gtf2 format. If takes and optional file_handle +# as the only argument to which it outputs the data. If no argument is +# given it outputs the data to stdout. +sub output_gtf_file{ + my ($gtf,$out_handle) = @_; + $gtf->_update; + unless(defined $out_handle){ + $out_handle = \*STDOUT; + } + if($gtf->comments){ + foreach my $comment (@{$gtf->comments}){ + print $out_handle "$comment\n"; + } + } + + for my $gene (@{$gtf->genes}) { $gene->_update } + my @top_level_features = sort {$a->start <=> $b->start} + (@{$gtf->genes}, @{$gtf->inter_cns}, @{$gtf->inter}); + for my $feature (@top_level_features) { + $feature->output_gtf($out_handle); + } +} + +# GTF::output_gff_file([file_handle]); +# This function moves through each gene in the gtf file and outputs +# all data in valid GFF format. If takes and optional file_handle +# as the only argument to which it outputs the data. If no argument is +# given it outputs the data to stdout. +sub output_gff_file{ + my ($gtf,$out_handle) = @_; + $gtf->_update; + unless(defined $out_handle){ + $out_handle = \*STDOUT; + } + if($gtf->genes){ + foreach my $gene (@{$gtf->genes}){ + $gene->output_gff($out_handle); + } + } +} + +# GTF::_parse_file() +# This function is used internally by the GTF constructor GTF::new +# to read, parse, and store the GTF file; +sub _parse_file{ + my ($gtf,$warnings) = @_; + unless(defined($warnings)){ + if(open(WARN, ">/dev/null")){ + $warnings = \*WARN; + } + else{ + die "Could not open /dev/null for write\n"; + } + } + my $fix_gtf = $gtf->{Fix_GTF}; + my $strict = $gtf->{Strict}; + #open file + open(GTF, "<$gtf->{Filename}") + or die "Could not open $gtf->{Filename}.\n"; + my (%GeneObj,%TxObj); + my (@all_genes,@all_txs,@all_cds,@all_inter,@all_cns); + my $all_line_comments = []; + #prepare error counts + my $max_error_count = 5; + if(defined($gtf->{Detailed_Error_Count})){ + $max_error_count = $gtf->{Detailed_Error_Count}; + } + my @error_msgs = $gtf->_get_error_messages(); + my @errors; + my @bad_list; + my $no_warn = $gtf->{Warning_Skips}; + for(my $i = 0;$i <= $#error_msgs;$i++){ + if($$no_warn[$i]){ + $errors[$i] = $max_error_count; + } + else{ + $errors[$i] = 0; + } + } + #read the file + my $line_num = 0; + while(my $in_line = ){ + $line_num++; + chomp $in_line; + if($in_line !~ /\S/){ + next; + } + if($in_line =~ /^\s*\#/){ + push @$all_line_comments, $in_line; + next; + } + else{ + #remove leading whitespace and get comments + my $comments = ""; + while($in_line =~ /^(.*)\#(.*)$/){ + $in_line = $1; + $comments = $2.$comments; + } + if($in_line =~ /^\s+(.*)$/){ + $in_line = $1; + } + my @data = split /\s+/,$in_line; + #verify line is correct length + if($#data < 8){ + if($errors[0] < $max_error_count){ + print $warnings + "Not enough fields on line $line_num.\n"; + } + $errors[0]++; + next; + } + #check for correct whitespace + my $field = 1; + while(($field < 8)&&($in_line =~ /\S+(\s+)/g)){ + unless($1 =~ /^\t$/){ + if($errors[1] < $max_error_count){ + print $warnings + "Incorrect type of whitespace between fields ", + "$field and ",$field + 1, " on line $line_num. ", + "Should be tab.\n"; + } + $errors[1]++; + } + $field++; + } + #verify correct field + $data[2] = lc($data[2]); + unless(($data[2] eq "cds") || ($data[2] eq "exon") + || ($data[2] eq "5utr") || ($data[2] eq "3utr") + || ($data[2] eq "start_codon") + || ($data[2] eq "stop_codon") + || ($data[2] eq "sec") + || ($data[2] eq "intron_cns") + || ($data[2] eq "inter_cns") + || ($data[2] eq "inter")) { + if($errors[2] < $max_error_count){ + print $warnings + "Incorrect value for field, \"$data[2]\", ", + "on line $line_num.\n"; + } + $errors[2]++; + } + if (($data[2] eq "cds") || ($data[2] eq "5utr") + || ($data[2] eq "3utr") || ($data[2] eq "sec")) { + $data[2] = uc($data[2]); + } + if ($data[2] eq "inter_cns") { $data[2] = "inter_CNS" } + if ($data[2] eq "intron_cns") { $data[2] = "intron_CNS" } + #verify correct format + if($data[3] !~ /^\d*$/){ + if($errors[3] < $max_error_count){ + print $warnings + "Non-numerical value for field, \"$data[3]\"", + ", on line $line_num.\n"; + } + $errors[3]++; + } + #verify correct format + if($data[4] !~ /^\d*$/){ + if($errors[4] < $max_error_count){ + print $warnings + "Non-numerical value for field, \"$data[4]\"", + ", on line $line_num.\n"; + } + $errors[4]++; + } + #verify start is < stop + if($data[3] > $data[4]){ + if($errors[38] < $max_error_count){ + print $warnings + "Start field is greater than stop field on line $line_num.\n"; + } + $errors[38]++; + my $start = $data[3]; + $data[3] = $data[4]; + $data[4] = $start; + } + #verify correct format + if($data[5] !~ /^-?\d*\.?\d*$/){ # should check for float + if($data[5] ne '.'){ + if($errors[5] < $max_error_count){ + print $warnings + "Value for field, \"$data[5]\", is ", + "neither numerical nor \".\" on line $line_num\n"; + } + $errors[5]++; + } + } + if(!$strict && $data[5] eq '.'){ + $data[5] = 0; + } + #verify correct field + unless(($data[6] eq "+")||($data[6] eq "-")|| + ($data[6] eq ".")){ + if($errors[6] < $max_error_count){ + print $warnings + "Value of field is \"$data[6]\" on line ", + "$line_num. Should be \"+\", \"-\", or \".\".\n"; + } + $errors[6]++; + if(!$strict){ + if($data[6] eq "1"){ + $data[6] = "+"; + } + elsif($data[6] eq "-1"){ + $data[6] = "-"; + } + elsif($data[6] eq "0"){ + $data[6] = "."; + } + } + } + #verify correct field + unless(($data[7] eq "0")||($data[7] eq "1")|| + ($data[7] eq "2")||($data[7] eq ".")){ + if($errors[7] < $max_error_count){ + print $warnings + "Value of field is \"$data[7]\" on line ", + "$line_num. Should be \"0\", \"1\", \"2\", or \".\"", + ".\n"; + } + $errors[7]++; + } + #prepare and fields + my $attribute_string = ""; + if($in_line =~ /^\S+\t\S+\t\S+\t\S+\t\S+\t\S+\t\S+\t\S+\t(.*)$/){ + $attribute_string = "$1 "; + } + #check for equals + if($attribute_string =~ / = /){ + if($errors[30] < $max_error_count){ + print $warnings + "Illegal \'=\' character in field on ", + "line $line_num\n"; + } + $errors[30]++; + $attribute_string =~ s/=/ /g; + } + # check for tab character. + if($attribute_string =~ /\t/){ + if($errors[9] < $max_error_count){ + print $warnings + "Illegal tab character in field on ", + "line $line_num.\n"; + } + $errors[9]++; + $attribute_string =~ s/\t/ /g; + } + # make sure it has semicolon terminator after each attribute + # otherwise fake it. + unless($attribute_string =~ /;/){ + if($errors[10] < $max_error_count){ + print $warnings + " field conatins no semicolons", + " terminating each attribute on line $line_num.\n"; + } + $errors[10]++; + my @att = split /\s+/, $attribute_string; + $attribute_string = ""; + my $att_count = $#att - (($#att+1) % 2); + for(my $i = 0;$i <= $att_count;$i+=2){ + $attribute_string .= $att[$i]." ".$att[$i+1]."; "; + } + } + # parse field + my $attributes = {}; + while($attribute_string =~ /(\S+)(\s+)(\S+)(\s*)/g){ + my $name = $1; + my $space1 = $2; + my $value = $3; + my $space2 = $4; + if($value =~ /^\"(\S*)\";$/){ + $value = $1; + } + else{ + if($value =~ /^(\S+);$/){ + $value = $1; + } + else{ + if($errors[10] < $max_error_count){ + $value =~ /(.)$/; + print $warnings + "Bad terminator, \"$1\", after ", + " name-value pair, $name $value, on ", + "line $line_num. Should be \";\".\n"; + } + $errors[10]++; + } + if($value =~ /^\"(\S+)\"$/){ + $value = $1; + } + else{ + if($errors[11] < $max_error_count){ + print $warnings + " field missing quotes around", + " values on line $line_num.\n"; + } + $errors[11]++; + } + } + $attributes->{$name} = $value; + unless($space1 =~ /^ $/){ + if($errors[12] < $max_error_count){ + print $warnings + "Bad separator, $space1, between ", + "name-value pair, $name $value, on line ", + "$line_num.\n"; + } + $errors[12]++; + } + unless($space2 =~ /^ $/){ + if($errors[12] < $max_error_count){ + print $warnings + "Bad separator, \"$space2\", between ", + " name-value pairs on line $line_num. Should be", + " \" \".\n$name $value\n"; + } + $errors[12]++; + } + } + # check for gene_id and transcript_id attributes + my $gid = "missing"; + if(defined($attributes->{gene_id})){ + $gid = $attributes->{gene_id}; + } + else{ + if($errors[13] < $max_error_count){ + print $warnings + "gene_id not set in field on line ", + "$line_num.\n"; + } + $errors[13]++; + $attributes->{gene_id} = "missing"; + } + my $tid = "missing"; + if(defined($attributes->{transcript_id})){ + $tid = $attributes->{transcript_id}; + } + else{ + if($errors[14] < $max_error_count){ + print $warnings + "transcript_id not set in field on line ", + "$line_num.\n"; + } + $errors[14]++; + $attributes->{transcript_id} = "missing"; + } + #create objects + my $feature = + GTF::Feature::new($data[2], $data[3], $data[4], $data[5], $data[7]); + if($feature->type eq "CDS"){ + push @all_cds, $feature; + } + elsif($feature->type eq "inter") { + push @all_inter, $feature; + } + elsif($feature->type eq "inter_CNS") { + push @all_cns, $feature; + } + + my $tx; + my $gene; + unless(defined($TxObj{$tid}) && length($tid)){ + $tx = GTF::Transcript::new($tid); + $TxObj{$tid} = $tx; + unless($feature->type eq "inter" || $feature->type eq "inter_CNS") { + push @all_txs, $tx; + } + unless(defined($GeneObj{$gid}) && length($gid)){ + $gene = GTF::Gene::new($gid,$data[0],$data[1],$data[6]); + $GeneObj{$gid} = $gene; + unless($feature->type eq "inter" || $feature->type eq "inter_CNS") { + push @all_genes, $gene; + } + } + $gene = $GeneObj{$gid}; + $gene->add_transcript($tx); + } + else{ + $gene = $GeneObj{$gid}; + } + $tx = $TxObj{$tid}; + #check that strand value is consistent with the strand of the tx + unless($data[6] eq $tx->strand){ + if($errors[20] < $max_error_count){ + print $warnings + "Inconsistent value across gene_id = ". + $tx->gene_id.".\n"; + } + $errors[20]++; + if (_check_errors_against_badlist($gtf,20)) { + push @bad_list, $tx->id; + } + if($strict){ + die; + } + #last;??? + } + $tx->add_feature($feature); + } + } + close(GTF); + #check each gene object + foreach my $tx (@all_txs){ + my $exons = $tx->exons; + my $cds = $tx->cds; + my $utr3 = $tx->utr3; + my $utr5 = $tx->utr5; + my $sec = $tx->sec; + my $inter = $tx->inter; + my $inter_cns = $tx->inter_cns; + if($#$exons >= 0 && $#$utr3 == -1 && $#$utr5 == -1){ + $tx->create_utr_objects_from_exons; + } + my $starts = $tx->start_codons; + my $stops = $tx->stop_codons; + my $strand = $tx->strand; + #make sure this tx contains at least one CDS, inter_CNS, or inter feature + if($#$cds == -1 && $#$inter_cns == -1 && $#$inter == -1){ + if($errors[15] < $max_error_count){ + print $warnings + "Transcript ".$tx->id." contains no CDS, inter or inter_CNS features.\n"; + } + $errors[15]++; + if (_check_errors_against_badlist($gtf,15)) { + push @bad_list, $tx->id; + } + next; + } + #verify start/stop codon exists and has length 3 + my $start_codon_start = -1; + my $start_codon_stop = -1; + if($#$starts == -1){ + if($errors[17] < $max_error_count){ + print $warnings + "Missing start_codon for transcript \"".$tx->id."\".\n"; + } + $errors[17]++; + if (_check_errors_against_badlist($gtf,17)) { + push @bad_list, $tx->id; + } + } + else{ + my $length = 0; + foreach my $start (@$starts){ + $length += $start->stop - $start->start + 1; + if(($start_codon_start < 0) || ($start_codon_start > $start->start)){ + $start_codon_start = $start->start; + } + if(($start_codon_stop < 0) || ($start_codon_stop < $start->stop)){ + $start_codon_stop = $start->stop; + } + } + if($length != 3){ + if($errors[16] < $max_error_count){ + print $warnings + "Start Codon length is not three in transcript \"".$tx->id."\".\n"; + } + $errors[16]++; + if (_check_errors_against_badlist($gtf,16)) { + push @bad_list, $tx->id; + } + } + } + my $stop_codon_start = -1; + my $stop_codon_stop = -1; + if($#$stops == -1){ + if($errors[19] < $max_error_count){ + print $warnings + "Missing stop_codon for transcript \"".$tx->id."\".\n"; + } + $errors[19]++; + if (_check_errors_against_badlist($gtf,19)) { + push @bad_list, $tx->id; + } + } + else{ + my $length = 0; + foreach my $stop (@$stops){ + $length += $stop->stop - $stop->start + 1; + if(($stop_codon_start < 0) || ($stop_codon_start > $stop->stop)){ + $stop_codon_start = $stop->start; + } + if(($stop_codon_stop < 0) || ($stop_codon_stop < $stop->stop)){ + $stop_codon_stop = $stop->stop; + } + } + if($length != 3){ + if($errors[18] < $max_error_count){ + print $warnings + "Stop Codon length is not three in gene \"".$tx->id."\".\n"; + } + $errors[18]++; + if (_check_errors_against_badlist($gtf,18)) { + push @bad_list, $tx->id; + } + } + } + + #check that no cds or utr features overlap each other + # also counts introns that are too short + # introns shorter than 4 bp are in all likelihood + # biologically impossible (at least 2 bp for each splice signal) + my $p_exons; + my $overlap = 0; + my $zero_intron_count = 0; + my $short_intron_count = 0; + foreach $p_exons ($utr5, $starts, $cds, $stops, $utr3) + { + my $last_stop; + if ($#$p_exons > 0) + { + $last_stop = $$p_exons[0]->stop; + } + + for(my $i = 1;$i <= $#$p_exons;$i++){ + my $intron_length = $$p_exons[$i]->start - $last_stop - 1; + if($intron_length < 0){ + $overlap++; + } + elsif($intron_length < 4){ + $zero_intron_count++; + } + elsif($intron_length < 20){ + $short_intron_count++; + } + $last_stop = $$p_exons[$i]->stop; + } + } + + # check CDS - UTR junctions for overlaps + # here, introns with length 0 are allowed because some + # exons could be in part UTR and in part CDS + if ($tx->strand eq "+") + { + my $ilen; + if (($#$utr5 > -1) && ($#$starts > -1)){ + $ilen = $$starts[0]->start - $$utr5[$#$utr5]->stop - 1; + $overlap++ if ($ilen < 0); + $zero_intron_count++ if (($ilen > 0) && ($ilen < 4)); + } + if (($#$cds > -1) && ($#$stops > -1)){ + # assumes stop codons are not included in CDS + $ilen = $$stops[0]->start - $$cds[$#$cds]->stop - 1; + $overlap++ if ($ilen < 0); + $zero_intron_count++ if (($ilen > 0) && ($ilen < 4)); + } + if (($#$stops > -1) && ($#$utr3 > -1)){ + $ilen = $$utr3[0]->start - $$stops[$#$stops]->stop - 1; + $overlap++ if ($ilen < 0); + $zero_intron_count++ if (($ilen > 0) && ($ilen < 4)); + } + } + else{ + my $ilen; + if (($#$utr3 > -1) && ($#$stops > -1)){ + $ilen = $$stops[0]->start - $$utr3[$#$utr3]->stop - 1; + $overlap++ if ($ilen < 0); + $zero_intron_count++ if (($ilen > 0) && ($ilen < 4)); + } + if (($#$stops > -1) && ($#$cds > -1)){ + # assumes stop codon is not included in CDS + $ilen = $$cds[0]->start - $$stops[$#$stops]->stop - 1; + $overlap++ if ($ilen < 0); + $zero_intron_count++ if (($ilen > 0) && ($ilen < 4)); + } + if (($#$starts > -1) && ($#$utr5 > -1)){ + $ilen = $$utr5[0]->start - $$starts[$#$starts]->stop - 1; + $overlap++ if ($ilen < 0); + $zero_intron_count++ if (($ilen > 0) && ($ilen < 4)); + } + } + + if($overlap > 0){ + if($errors[45] < $max_error_count){ + print $warnings + "Overlapping CDS or UTR features in transcript \"".$tx->id."\".\n"; + } + $errors[45]++; + if (_check_errors_against_badlist($gtf,45)) { + push @bad_list, $tx->id; + } + } + if($zero_intron_count > 0){ + if($errors[48] < $max_error_count){ + print $warnings "$zero_intron_count intron(s) shorter than 4 bp found on in transcript \"".$tx->id."\".\n" ; + } + $errors[48]++; + if (_check_errors_against_badlist($gtf,48)) { + push @bad_list, $tx->id; + } + } + if($short_intron_count > 0){ + if($errors[49] < $max_error_count){ + print $warnings "$short_intron_count short (<20bp) intron(s) found on in transcript \"".$tx->id."\".\n" ; + } + $errors[49]++; + if (_check_errors_against_badlist($gtf,49)) { + push @bad_list, $tx->id; + } + } + #check that each cds is between the start and stop and that the initial and + #terminal exons are placed correctly with respect to the start/stop codons + if($start_codon_start >= 0){ + if($strand eq '-'){ + for(my $i = $#$cds;$i >= 0;$i--){ + if($$cds[$i]->end > $start_codon_stop){ + if($errors[21] < $max_error_count){ + print $warnings + "CDS before start codon in transcript_id = \"". + "".$tx->id."\".\n"; + } + $errors[21]++; + if (_check_errors_against_badlist($gtf,21)) { + push @bad_list, $tx->id; + } + } + } + my $cds = $tx->initial_exon; + if($cds->end != $start_codon_stop){ + if($errors[33] < $max_error_count){ + print $warnings + "Start codon location not consistent with initial exon ". + "locaiton in transcript_id = \"".$tx->id."\".\n"; + } + $errors[33]++; + if (_check_errors_against_badlist($gtf,33)) { + push @bad_list, $tx->id; + } + } + if($cds->length < 3){ + my $cds = $tx->cds; + my $error = 0; + for(my $i = 0;$i <= $#$starts;$i++){ + if($i == $#$starts){ + unless($$cds[$#$cds-$i]->start <= $$starts[$#$starts-$i]->start){ + $error = 1; + } + } + else{ + unless($$cds[$#$cds-$i]->start == $$starts[$#$starts-$i]->start){ + $error = 1; + } + } + } + if($error){ + if($errors[47] < $max_error_count){ + print $warnings "Start codon annotated in intron region on ". + "transcript_id = \"".$tx->id."\".\n"; + } + $errors[47]++; + if (_check_errors_against_badlist($gtf,47)) { + push @bad_list, $tx->id; + } + my $pos = 0; + my $start_pos = 0; + my $cds_pos = $#$cds; + #fix the start codon by splitting it across the cds + while($pos < 3){ + my $end; + if($$cds[$cds_pos]->length < 3-$pos){ + $end = $$cds[$cds_pos]->start; + $pos += $$cds[$cds_pos]->length; + } + else{ + $end = $$cds[$cds_pos]->end-2+$pos; + $pos = 3; + } + if($start_pos <= $#$starts){ + $$starts[$start_pos]->set_start($end); + $$starts[$start_pos]->set_stop($$cds[$cds_pos]->end); + } + else{ + my $sc = GTF::Feature::new("start_codon", $end, $$cds[$cds_pos]->end, ".", "."); + $tx->add_feature($sc); + } + $cds_pos--; + $start_pos++; + } + $starts = [sort {$a->start <=> $b->start} @$starts]; + } + } + } + else{ + for(my $i = 0;$i <= $#$cds;$i++){ + if($$cds[$i]->start < $start_codon_start){ + if($errors[21] < $max_error_count){ + print $warnings + "CDS before start codon in transcript_id = \"". + "".$tx->id."\".\n"; + } + $errors[21]++; + if (_check_errors_against_badlist($gtf,21)) { + push @bad_list, $tx->id; + } + } + } + my $cds = $tx->initial_exon; + if($cds->start != $start_codon_start){ + if($errors[33] < $max_error_count){ + print $warnings + "Start codon location not consistent with initial exon ". + "locaiton in transcript_id = \"".$tx->id."\".\n"; + } + $errors[33]++; + if (_check_errors_against_badlist($gtf,33)) { + push @bad_list, $tx->id; + } + } + if($cds->length < 3){ + my $cds = $tx->cds; + my $error = 0; + for(my $i = 0; $i <= $#$starts;$i++){ + if($i == $#$starts){ + unless($$cds[$i]->stop >= $$starts[$i]->stop){ + $error = 1; + } + } + else{ + unless($$cds[$i]->stop == $$starts[$i]->stop){ + $error = 1; + } + } + } + if($error){ + if($errors[47] < $max_error_count){ + print $warnings "Start codon annotated in intron region on ". + "transcript_id = \"".$tx->id."\".\n"; + } + $errors[47]++; + if (_check_errors_against_badlist($gtf,47)) { + push @bad_list, $tx->id; + } + my $pos = 0; + my $start_pos = 0; + my $cds_pos = 0; + #fix the start codon by splitting it across the cds + while($pos < 3){ + my $end; + if($$cds[$cds_pos]->length < 3-$pos){ + $end = $$cds[$cds_pos]->stop; + $pos += $$cds[$cds_pos]->length; + } + else{ + $end = $$cds[$cds_pos]->start+2-$pos; + $pos = 3; + } + if($start_pos <= $#$starts){ + $$starts[$start_pos]->set_start($$cds[$cds_pos]->start); + $$starts[$start_pos]->set_stop($end); + } + else{ + my $sc = GTF::Feature::new("start_codon", $$cds[$cds_pos]->start, $end, ".", "."); + $tx->add_feature($sc); + } + $cds_pos++; + $start_pos++; + } + $starts = [sort {$a->start <=> $b->start} @$starts]; + } + } + } + } + if($stop_codon_start >= 0){ + if($strand =~ /\-/){ + for(my $i = 0;$i <= $#$cds;$i++){ + if($$cds[$i]->start < $stop_codon_start){ + if($errors[22] < $max_error_count){ + print $warnings + "CDS after stop codon in transcript_id = \"". + "".$tx->id."\".\n"; + } + $errors[22]++; + if (_check_errors_against_badlist($gtf,22)) { + push @bad_list, $tx->id; + } + } + } + my $cds = $tx->terminal_exon; + if($cds->start == $stop_codon_start){ + unless($cds->stop < $stop_codon_stop + 1){ + $cds->set_start($stop_codon_stop + 1); + } + if($errors[46] < $max_error_count){ + print $warnings + "Stop codon included in terminal CDS transcript_id = ". + "\"".$tx->id."\".\n"; + } + $errors[46]++; + if (_check_errors_against_badlist($gtf,46)) { + push @bad_list, $tx->id; + } + } + if($cds->start != $stop_codon_stop + 1){ + if($errors[34] < $max_error_count){ + print $warnings + "Stop codon location not consistent with terminal exon ". + "locaiton in transcript_id = \"".$tx->id."\".\n"; + } + $errors[34]++; + if (_check_errors_against_badlist($gtf,34)) { + push @bad_list, $tx->id; + } + + } + } + else{ + for(my $i = $#$cds;$i >= 0;$i--){ + if($$cds[$i]->end > $stop_codon_stop){ + if($errors[22] < $max_error_count){ + print $warnings + "CDS after stop codon in transcript_id = \"".$tx->id."\".\n"; + } + $errors[22]++; + if (_check_errors_against_badlist($gtf,22)) { + push @bad_list, $tx->id; + } + } + } + my $cds = $tx->terminal_exon; + if($cds->stop == $stop_codon_stop){ + unless($cds->start > $stop_codon_start -1){ + $cds->set_stop($stop_codon_start - 1); + } + if($errors[46] < $max_error_count){ + print $warnings + "Stop codon included in terminal CDS transcript_id = ". + "\"".$tx->id."\".\n"; + } + $errors[46]++; + if (_check_errors_against_badlist($gtf,46)) { + push @bad_list, $tx->id; + } + } + if($cds->stop != $stop_codon_start - 1){ + if($errors[34] < $max_error_count){ + print $warnings + "Stop codon location not consistent with terminal exon ". + "locaiton in transcript_id = \"".$tx->id."\".\n"; + } + $errors[34]++; + if (_check_errors_against_badlist($gtf,34)) { + push @bad_list, $tx->id; + } + } + } + } + #check that complete transcripts have even number of codons + if(($start_codon_start >= 0) && ($stop_codon_stop >= 0)){ + my $len = 0; + for(my $i = $#$cds;$i >= 0;$i--){ + $len += $$cds[$i]->length; + } + unless(($len % 3) == 0){ + if($errors[41] < $max_error_count){ + print $warnings + "Complete transcript length not a multiple of three". + " on transcript_id = \"".$tx->id."\".\n"; + } + $errors[41]++; + if (_check_errors_against_badlist($gtf,41)) { + push @bad_list, $tx->id; + } + } + } + #check for consistent phase + my $phase = -1; + if($start_codon_start >= 0){ + $phase = 0; + } + elsif($start_codon_stop >= 0){ + my $len = 0; + for(my $i = $#$cds;$i >= 0;$i--){ + $len += $$cds[$i]->length; + } + $phase = $len % 3; + } + if($strand eq "-"){ + if($phase == -1){ + $phase = $$cds[$#$cds]->frame; + if($phase eq "."){ + my $pos = -1; + for(my $i = $#$cds; $i >= 0;$i--){ + unless($$cds[$i]->frame eq "."){ + $pos = $i; + $phase = $$cds[$i]->frame; + last; + } + } + if($pos >= 0){ + for(my $i = $pos + 1; $i <= $#$cds;$i++){ + my $len = $$cds[$i]->length + $phase; + $phase = $len % 3; + } + } + } + + } + unless($phase eq "."){ + my $bad_frame = 0; + for(my $i = $#$cds;$i >= 0;$i--){ + unless($phase eq $$cds[$i]->frame){ + unless($$cds[$i]->frame eq "."){ + $bad_frame = 1; + } + $$cds[$i]->set_frame($phase); + } + $phase = ($$cds[$i]->length - $phase) % 3; + if($phase == 1){ + $phase = 2; + } + elsif($phase == 2){ + $phase = 1; + } + } + if($bad_frame){ + if($errors[42] < $max_error_count){ + print $warnings + "Inconsistent frame accross transcript_id = ". + "\"".$tx->id."\".\n"; + } + $errors[42]++; + if (_check_errors_against_badlist($gtf,42)) { + push @bad_list, $tx->id; + } + } + } + } + else{ + if($phase == -1){ + $phase = $$cds[0]->frame; + if($phase eq "."){ + my $pos = -1; + for(my $i = 0; $i <= $#$cds;$i++){ + unless($$cds[$i]->frame eq "."){ + $pos = $i; + $phase = $$cds[$i]->frame; + last; + } + } + if($pos >= 0){ + for(my $i = $pos - 1; $i >= 0;$i--){ + my $len = $$cds[$i]->length + $phase; + $phase = $len % 3; + } + } + } + } + unless($phase eq "."){ + my $bad_frame = 0; + for(my $i = 0; $i <= $#$cds;$i++){ + unless($phase eq $$cds[$i]->frame){ + unless($$cds[$i]->frame eq "."){ + $bad_frame = 1; + } + $$cds[$i]->set_frame($phase); + } + $phase = ($$cds[$i]->length - $phase) % 3; + if($phase == 1){ + $phase = 2; + } + elsif($phase == 2){ + $phase = 1; + } + } + if($bad_frame){ + if($errors[42] < $max_error_count){ + print $warnings + "Inconsistent frame accross transcript_id = ". + "\"".$tx->id."\".\n"; + } + $errors[42]++; + if (_check_errors_against_badlist($gtf,42)) { + push @bad_list, $tx->id; + } + } + } + } + # check selenocysteine (SEC) features here. must be inside a + # CDS and have frame 0 + if($#$sec >= 0){ + foreach my $s (@$sec){ + if($s->frame ne "0" && $s->frame ne "."){ + #bad SEC frame + if($errors[51] < $max_error_count){ + print $warnings + "Bad frame, ".$s->frame."on SEC feature on tx ".$tx->id.". Must of 0 or \".\"\n"; + } + $errors[51]++; + if (_check_errors_against_badlist($gtf,51)) { + push @bad_list, $tx->id; + } + } + my $good = 0; + foreach my $c (@$cds){ + if($c->stop < $s->start){ + next; + } + if($c->start <= $s->start && + $c->stop >= $s->stop){ + $good = 1; + last; + } + if($c->start > $s->stop){ + last; + } + } + if($good == 0){ + # SEC outside CDS + if($errors[52] < $max_error_count){ + print $warnings + "SEC feature outside of CDS region on tx ".$tx->id.".\n"; + } + $errors[52]++; + if (_check_errors_against_badlist($gtf,52)) { + push @bad_list, $tx->id; + } + + } + } + } + } + + #if the conservation sequence is available store it's info; + my $conseqname = $gtf->{Conseq}; + my %total_conserv = (0 => 0,1 => 0,2 => 0); + if($conseqname){ + open(Conseq, $conseqname) + or die "Could not open conservation sequence file: \"$conseqname\".\n"; + my @conseq_info = stat(Conseq); + my @comp; + foreach my $tx (@all_txs){ + my $cds = $tx->cds; + my $exons = $tx->exons; + my $starts = $tx->start_codons; + my $stops = $tx->stop_codons; + foreach my $start (@$starts){ + push @comp,{feature => $start, "0" => 0, "1" => 0, "2" => 0}; + } + foreach my $stop (@$stops){ + push @comp,{feature => $stop, "0" => 0, "1" => 0, "2" => 0}; + } + foreach my $exon (@$cds){ + push @comp,{feature => $exon, "0" => 0, "1" => 0, "2" => 0}; + } + foreach my $exon (@$exons){ + push @comp,{feature => $exon, "0" => 0, "1" => 0, "2" => 0}; + } + } + @comp = sort {$a->{feature}->start <=> $b->{feature}->start} @comp; + my $base = "X"; + my $nuc_pos = 0; + my $file_pos = 0; + my $line = "X"; + my $next_comp = 0; + my $max_comps = $#comp; + my %checking; + while(($next_comp <= $max_comps) and ($file_pos < $conseq_info[7])){ + $base = getc(Conseq); + unless(defined($base)){ + print $warnings "Features beyond end of conservation sequence.\n"; + last; + } + $file_pos++; + if($base =~ /\d/){ + $nuc_pos++; + if(defined($total_conserv{$base})){ + $total_conserv{$base}++; + } + else{ + $total_conserv{X}++; + if($errors[43] < $max_error_count){ + print $warnings "Bad conservation sequence character, $base. ". + "Should be \'0\',\'1\', or \'2\'.\n"; + } + $errors[43]++; + } + #compile composition stats + while(($next_comp <= $max_comps) and + ($nuc_pos == $comp[$next_comp]->{feature}->start)){ + $checking{$comp[$next_comp]} = $comp[$next_comp]; + $next_comp++; + } + foreach my $id (keys %checking){ + $checking{$id}->{$base}++; + if($checking{$id}->{feature}->stop == $nuc_pos){ + delete $checking{$id}; + } + } + } + } + while($file_pos < $conseq_info[7]){ + $base = getc(Conseq); + $file_pos++; + if($base =~ /\d/){ + $nuc_pos++; + if(defined($total_conserv{$base})){ + $total_conserv{$base}++; + } + else{ + $total_conserv{X}++; + if($errors[43] < $max_error_count){ + print $warnings "Bad conservation sequence character, $base. ". + "Should be \'0\',\'1\', or \'2\'.\n"; + } + $errors[43]++; + } + } + } + close(Conseq); + foreach my $info (@comp){ + $info->{feature}->set_conseq($info->{0},$info->{1},$info->{2}); + } + } + $gtf->{Total_Conseq} = \%total_conserv; + #if the sequence is available use it to fix any problems + my @inframe_stops; + my $seqname = $gtf->{Sequence}; + my %tx_phase; + foreach my $tx (@all_txs){ + $tx_phase{$tx->id} = -1; + } + my %total_seq = ('A' => 0,'C' => 0,'G' =>0,'T' => 0,'N' => 0,'X' => 0); + #EVAN need to rewrite all this sequence loading stuff because it is fugly + if($seqname){ + open(SEQ, "<$seqname") + or die "Could not open sequence file: \"$seqname\".\n"; + my @seq_info = stat(SEQ); + #try to fix things with the sequence info here + #check start codons, stop codons, splice site, internal stop + my %sequence; + my %seq_pos; + my @checks; + foreach my $tx (@all_txs){ + $sequence{$tx->id} = ""; + $seq_pos{$tx->id} = 0; + my $starts = $tx->start_codons; + my $stops = $tx->stop_codons; + my $utr3 = $tx->utr3; + my $utr5 = $tx->utr5; + my $cds = $tx->cds; + if($#$cds == -1){ + next; + } + my $exons = $tx->exons; + foreach my $start (@$starts){ + push @checks, {pos => $start->start, + type => "start_sequence", + tx => $tx, + object => $start, + stop => $start->stop}; + } + foreach my $stop (@$stops){ + push @checks, {pos => $stop->start, + type => "stop_sequence", + tx => $tx, + object => $stop, + stop => $stop->stop}; + } + foreach my $exon (@$exons){ + push @checks, {pos => $exon->start, + type => "exon_sequence", + tx => $tx, + object => $exon, + stop => $exon->stop}; + } + foreach my $coding (@$cds){ + push @checks, {pos => $coding->start, + type => "coding_sequence", + tx => $tx, + object => $coding, + stop => $coding->stop}; + } + + my($p_exons, $splice_type1, $splice_type2); + if ($tx->strand eq "+"){ + $splice_type1 = "donor_splice"; + $splice_type2 = "acceptor_splice"; + } + else{ + $splice_type1 = "acceptor_splice"; + $splice_type2 = "donor_splice"; + } + + foreach $p_exons ($utr5, $starts, $cds, $stops, $utr3){ + next if ($#$p_exons < 1); + + my($i, $subtype); + $subtype = ($$p_exons[0]->type =~ /UTR/) ? "utr" : "cds"; + for ($i = 1; $i <= $#$p_exons; $i++){ + push @checks, {pos => $$p_exons[$i-1]->stop+2, + type => $subtype."_".$splice_type1, + tx => $tx, + stop => 0}; + push @checks, {pos => $$p_exons[$i]->start-1, + type => $subtype."_".$splice_type2, + tx => $tx, + stop => 0}; + } + } + + # this assumes that the innermost start codon is included in CDS + # therefore, no checks are run on start_codon - CDS junction + # also, this allows zero-length introns at UTR5 - start_codon + # CDS - stop_codon and UTR3 - stop_codon junctions + if ($tx->strand eq "+"){ + if (($#$utr5 > -1) && ($#$starts > -1)){ + my $ilen = $$starts[0]->start - $$utr5[$#$utr5]->stop - 1; + if ($ilen != 0){ + push @checks, {pos => $$utr5[$#$utr5]->stop+2, + type => "utr_donor_splice", + tx => $tx, + stop => 0}; + push @checks, {pos => $$starts[0]->start-1, + type => "utr_acceptor_splice", + tx => $tx, + stop => 0}; + } + } + if (($#$cds > -1) && ($#$stops > -1)){ + my $ilen = $$stops[0]->start - $$cds[$#$cds]->stop - 1; + if ($ilen != 0){ + push @checks, {pos => $$cds[$#$cds]->stop+2, + type => "cds_donor_splice", + tx => $tx, + stop => 0}; + push @checks, {pos => $$stops[0]->start-1, + type => "cds_acceptor_splice", + tx => $tx, + stop => 0}; + } + } + if (($#$stops > -1) && ($#$utr3 > -1)){ + my $ilen = $$utr3[0]->start - $$stops[$#$stops]->stop - 1; + if ($ilen != 0){ + push @checks, {pos => $$stops[$#$stops]->stop+2, + type => "utr_donor_splice", + tx => $tx, + stop => 0}; + push @checks, {pos => $$utr3[0]->start-1, + type => "utr_acceptor_splice", + tx => $tx, + stop => 0}; + } + } + if ($#{$stops} < 0){ + push @checks, {pos => $$cds[$#{$cds}]->stop+1, + type => "stop_sequence?", + tx => $tx, + stop => $$cds[$#{$cds}]->stop+3}; + } + if ($#{$starts} < 0){ + if ($$cds[0]->start-3 > 0){ + push @checks, {pos => $$cds[0]->start-3, + type => "start_sequence?", + tx => $tx, + stop => $$cds[0]->start-1}; + } + else{ + $sequence{$tx->id} = "NNN"; + } + } + } + else{ + if (($#$utr3 > -1) && ($#$stops > -1)){ + my $ilen = $$stops[0]->start - $$utr3[$#$utr3]->stop - 1; + if ($ilen != 0){ + push @checks, {pos => $$utr3[$#$utr3]->stop+2, + type => "utr_acceptor_splice", + tx => $tx, + stop => 0}; + push @checks, {pos => $$stops[0]->start-1, + type => "utr_donor_splice", + tx => $tx, + stop => 0}; + } + } + if (($#$stops > -1) && ($#$cds > -1)){ + my $ilen = $$cds[0]->start - $$stops[$#$stops]->stop - 1; + if ($ilen != 0){ + push @checks, {pos => $$stops[$#$stops]->stop+2, + type => "cds_acceptor_splice", + tx => $tx, + stop => 0}; + push @checks, {pos => $$cds[0]->start-1, + type => "cds_donor_splice", + tx => $tx, + stop => 0}; + } + } + if (($#$starts > -1) && ($#$utr5 > -1)){ + my $ilen = $$utr5[0]->start - $$starts[$#$starts]->stop - 1; + if ($ilen != 0){ + push @checks, {pos => $$starts[$#$starts]->stop+2, + type => "utr_acceptor_splice", + tx => $tx, + stop => 0}; + push @checks, {pos => $$utr5[0]->start-1, + type => "utr_donor_splice", + tx => $tx, + stop => 0}; + } + } + if ($#{$stops} < 0){ + if ($$cds[0]->start-3 > 0){ + push @checks, {pos => $$cds[0]->start-3, + type => "stop_sequence?", + tx => $tx, + stop => $$cds[0]->start-1}; + } + } + if ($#{$starts} < 0){ + push @checks, {pos => $$cds[$#{$cds}]->stop+1, + type => "start_sequence?", + tx => $tx, + stop => $$cds[$#{$cds}]->stop+3}; + } + } + } + + @checks = sort{$a->{pos} <=> $b->{pos} || $a->{stop} <=> $b->{stop}} @checks; + my $file_pos = length(); + my $base = "X"; + my $nuc_pos = 0; + my $line = "X"; + my $next = 0; + my $prev_base = "X"; + my $max_checks = $#checks; + my %checking; + my $splice_type; + + while((($next <= $max_checks) || ((keys %checking) >= 0)) + && ($file_pos < $seq_info[7])){ + $base = getc(SEQ); + unless(defined($base)){ + print $warnings "Features beyond end of sequence.\n"; + last; + } + $file_pos++; + if($base =~ /\S/){ + $base = uc($base); + if((defined($total_seq{$base})) && ($base ne 'X')){ + $total_seq{$base}++; + } + else{ + $total_seq{X}++; + if($errors[44] < $max_error_count){ + print $warnings "Bad fasta sequence character, $base. Should be". + " \'A\',\'C\',\'T\',\'G\',or \'N\'.\n"; + } + $errors[44]++; + } + $nuc_pos++; + while(($next <= $max_checks) && ($checks[$next]{pos} == $nuc_pos)){ + if($checks[$next]{type} =~ /sequence/){ + $checking{$checks[$next]{tx}->id . "\t" . + $checks[$next]{type}} = { + check => $checks[$next], + A => 0, + C => 0, + G => 0, + T => 0, + N => 0}; + } + elsif(($checks[$next]{type} =~ /^(\w+)_acceptor_splice$/) + && ($splice_type = $1)) { + if($checks[$next]{tx}->strand eq "-"){ + unless(($prev_base eq "C") && ($base eq "T")){ + if (($prev_base eq "G") && ($base eq "T")) { + if ($errors[50] < $max_error_count) { + print $warnings + "Possible non-canonical splice site sequence, ". + "$prev_base$base, on negative strand transcript \"". + $checks[$next]{tx}->id. + "\".\n"; + } + $errors[50]++; + if ((_check_errors_against_badlist($gtf,50)) + && ($splice_type ne "utr")) { + push @bad_list, $checks[$next]{tx}->id; + } + } + else { + if($errors[26] < $max_error_count){ + print $warnings + "Bad acceptor splice site sequence, $prev_base". + "$base, on negative strand transcript \"". + $checks[$next]{tx}->id. + "\". Should be CT or GT.\n"; + } + $errors[26]++; + if (_check_errors_against_badlist($gtf,26)) { + push @bad_list, $checks[$next]{tx}->id; + } + } + } + + } + else{ + unless(($prev_base eq "A") && ($base eq "G")){ + if (($prev_base eq "A") && ($base eq "C")) { + if ($errors[50] < $max_error_count) { + print $warnings + "Possible non-canonical splice site sequence, ". + "$prev_base$base, on transcript \"". + $checks[$next]{tx}->id. + "\".\n"; + } + $errors[50]++; + if ((_check_errors_against_badlist($gtf,50)) + && ($splice_type ne "utr")) { + push @bad_list, $checks[$next]{tx}->id; + } + } + else { + if($errors[26] < $max_error_count){ + print $warnings + "Bad acceptor splice site sequence, $prev_base". + "$base, on transcript \"". + $checks[$next]{tx}->id. + "\". Should be AG or AC.\n"; + } + $errors[26]++; + if (_check_errors_against_badlist($gtf,26)) { + push @bad_list, $checks[$next]{tx}->id; + } + } + } + } + } + elsif(($checks[$next]{type} =~ /^(\w+)_donor_splice$/) + && ($splice_type = $1)) { + if($checks[$next]{tx}->strand eq "-"){ + unless(($prev_base eq "A") && ($base eq "C")){ + if (($prev_base eq "A") && ($base eq "T")) { + if ($errors[50] < $max_error_count) { + print $warnings + "Possible non-canonical splice site sequence, ". + "$prev_base$base, on negative strand transcript \"". + $checks[$next]{tx}->id. + "\".\n"; + } + $errors[50]++; + if ((_check_errors_against_badlist($gtf,50)) + && ($splice_type ne "utr")) { + push @bad_list, $checks[$next]{tx}->id; + } + } + elsif (($prev_base eq "G") && ($base eq "C")) { + if ($errors[50] < $max_error_count) { + print $warnings + "Possible non-canonical splice site sequence, ". + "$prev_base$base, on negative strand transcript \"". + $checks[$next]{tx}->id. + "\".\n"; + } + $errors[50]++; + if ((_check_errors_against_badlist($gtf,50)) + && ($splice_type ne "utr")) { + push @bad_list, $checks[$next]{tx}->id; + } + } + else { + if ($errors[27] < $max_error_count){ + print $warnings + "Bad donor splice site sequence, $prev_base". + "$base, on negative strand transcript \"". + $checks[$next]{tx}->id. + "\". Should be AC, GC, or AT.\n"; + } + $errors[27]++; + if (_check_errors_against_badlist($gtf,27)) { + push @bad_list, $checks[$next]{tx}->id; + } + } + } + } + else{ + unless(($prev_base eq "G") && ($base eq "T")){ + if (($prev_base eq "G") && ($base eq "C")) { + if ($errors[50] < $max_error_count) { + print $warnings + "Possible non-canonical splice site sequence, ". + "$prev_base$base, on transcript \"". + $checks[$next]{tx}->id. + "\".\n"; + } + $errors[50]++; + if ((_check_errors_against_badlist($gtf,50)) + && ($splice_type ne "utr")) { + push @bad_list, $checks[$next]{tx}->id; + } + } + elsif (($prev_base eq "A") && ($base eq "T")) { + if ($errors[50] < $max_error_count) { + print $warnings + "Possible non-canonical splice site sequence, ". + "$prev_base$base, on transcript \"". + $checks[$next]{tx}->id. + "\".\n"; + } + $errors[50]++; + if ((_check_errors_against_badlist($gtf,50)) + && ($splice_type ne "utr")) { + push @bad_list, $checks[$next]{tx}->id; + } + } + else { + if ($errors[27] < $max_error_count){ + print $warnings + "Bad donor splice site sequence, $prev_base". + "$base, on transcript \"". + $checks[$next]{tx}->id. + "\". Should be GT, GC, or AT.\n"; + } + $errors[27]++; + if (_check_errors_against_badlist($gtf,27)) { + push @bad_list, $checks[$next]{tx}->id; + } + } + } + } + } + $next++; + } + foreach my $check (keys %checking){ + $check =~ /^(.+)\t.+$/; + my $tx_id = $1; + unless(($checking{$check}{check}{type} =~ /exon/) || + ($seq_pos{$tx_id} >= $nuc_pos)){ + $sequence{$tx_id}.= uc($base); + $seq_pos{$tx_id} = $nuc_pos; + } + $checking{$check}{uc($base)}++; + if($checking{$check}{check}{stop} eq $nuc_pos){ + my $object = $checking{$check}{check}{object}; + if(defined($object)){ + if($object->strand eq "-"){ + $object->set_bases($checking{$check}{T}, + $checking{$check}{G}, + $checking{$check}{C}, + $checking{$check}{A}, + $checking{$check}{N}); + + } + else{ + $object->set_bases($checking{$check}{A}, + $checking{$check}{C}, + $checking{$check}{G}, + $checking{$check}{T}, + $checking{$check}{N}); + } + } + delete $checking{$check}; + } + } + $prev_base = $base; + } + } + foreach my $tx (@all_txs){ + my $cds = $tx->cds; + unless($#$cds >= 0){ + next; + } + my $sec = $tx->sec; + my $starts = $tx->start_codons; + my $stops = $tx->stop_codons; + my $strand = $tx->strand; + my $tx_id = $tx->id; + my $phase = $$cds[0]->frame; + my $old_phase = $phase; + if($strand eq "-"){ + $phase = $$cds[$#$cds]->frame; + } + my $seq = $sequence{$tx->id}; + if($seq =~ /^$/){ + print STDERR "No sequence for $tx_id\n"; + next; + } + if($tx->strand eq "-"){ + $seq = _rev_comp($seq); + } + my $start = 0; + my $stop = 0; + my $found_start = -1; + my $found_stop = -1; + my @seq = split //, $seq; + #check some SEC stuff here; + foreach my $s (@$sec){ + my $pos = 0; + if($tx->strand eq '-'){ + foreach my $c (reverse @$cds){ + if($c->start > $s->stop){ + $pos += $c->stop-$c->start+1; + } + else{ + $pos += $c->stop-$s->stop; + last; + } + } + if($#$stops == -1){ + $pos += 3; + } + } + else{ + foreach my $c (@$cds){ + if($c->stop < $s->start){ + $pos += $c->stop-$c->start+1; + } + else{ + $pos += $s->start-$c->start; + last; + } + } + if($#$starts == -1){ + $pos += 3; + } + } + my $codon = substr($seq,$pos,3); + unless($codon eq "TGA"){ + #bad SEC codon + $error_msgs[53] = "SEC feature had bad sequence. Must be \"TGA\""; + if($errors[53] < $max_error_count){ + print $warnings "SEC feature had bad sequence, $codon, on tx, ".$tx->id. + ". Must be \"TGA\"\n"; + } + $errors[53]++; + if (_check_errors_against_badlist($gtf,53)) { + push @bad_list, $tx->id; + } + } + substr($seq,$pos,3) = "NNN"; + } + if($#$starts >= 0){ + my $sc = substr($seq,0,3); + unless($sc eq "ATG"){ + if($errors[24] < $max_error_count){ + print $warnings + "Bad start codon sequence, $sc, on tx ". + $tx->id.". Should be ATG.\n"; + } + $errors[24]++; + if (_check_errors_against_badlist($gtf,24)) { + push @bad_list, $tx->id; + } + } + } + else{ + my $sc1 = substr($seq,0,3); + my $sc2 = substr($seq,3,3); + if($sc2 eq "ATG"){ + $found_start = 3; + $start = 3; + } + elsif($sc1 eq "ATG"){ + $found_start = 0; + } + else{ + $start = 3; + } + } + if($#$stops >= 0){ + my $sc = substr($seq,$#seq-2,3); + unless(($sc eq "TAA") || + ($sc eq "TAG") || + ($sc eq "TGA")){ + if($errors[25] < $max_error_count){ + print $warnings + "Bad stop codon sequence, $sc, on tx ". + $tx->id.". Should be TAA, TAG, or TGA.\n"; + } + $errors[25]++; + if (_check_errors_against_badlist($gtf,25)) { + push @bad_list, $tx->id; + } + } + } + else{ + my $sc1 = substr($seq,$#seq - 2,3); + my $sc2 = substr($seq,$#seq - 5,3); + if(($sc2 eq "TAA") || + ($sc2 eq "TAG") || + ($sc2 eq "TGA")){ + $found_stop = $#seq - 5; + $stop = 3; + } + elsif(($sc1 eq "TAA") || + ($sc1 eq "TAG") || + ($sc1 eq "TGA")){ + $found_stop = $#seq - 2; + } + else{ + $stop = 3; + } + } + my @sc = (0,0,0); + my $scphase = $phase; + $scphase = 0 if($scphase eq '.'); + for(my $i = $start;$i <= $#seq - $stop - 5;$i++){ + my $codon = substr($seq,$i,3); + if(($codon eq "TAA") || + ($codon eq "TAG") || + ($codon eq "TGA")){ + $sc[($i+$scphase) % 3] = 1; + } + } + if(($phase ne ".") && + ($sc[0])){ + if($errors[29] < $max_error_count){ + print $warnings "In frame stop codon(s) found on transcript ". + $tx->id."\n"; + } + $errors[29]++; + if (_check_errors_against_badlist($gtf,29)) { + push @bad_list, $tx->id; + } + push @inframe_stops, $tx->id; + } + else{ + if(($sc[0] == 1) && + ($sc[1] == 1) && + ($sc[2] == 1)){ + push @inframe_stops, $tx->id; + if($errors[37] < $max_error_count){ + print $warnings "Stop Codons in all frames on transcript ". + $tx->id."\n"; + } + $errors[37]++; + if (_check_errors_against_badlist($gtf,37)) { + push @bad_list, $tx->id; + } + } + } + # add start or stop if possible + if($found_stop != -1){ + if(($phase eq ".") || + ($phase == (($#seq + 1) % 3))){ + if($sc[($#seq +1) % 3] == 0){ + $phase = ($#seq + 1) % 3; + if($errors[32] < $max_error_count){ + print $warnings + "Unannotated stop codon found on transcript ". + $tx->id."\n"; + } + $errors[32]++; + if (_check_errors_against_badlist($gtf,32)) { + push @bad_list, $tx->id; + } + if($fix_gtf){ + if($strand eq "+"){ + my $terminal_exon = $$cds[$#$cds]; + my $sc_start = $terminal_exon->stop + 1 - $stop; + my $stop_codon = + GTF::Feature::new("stop_codon",$sc_start, + $sc_start + 2,0,0); + $terminal_exon->set_stop($sc_start - 1); + $tx->add_feature($stop_codon); + } + else{ + my $terminal_exon = $$cds[0]; + my $sc_start = $terminal_exon->start - 3 + $stop; + my $stop_codon = + GTF::Feature::new("stop_codon",$sc_start, + $sc_start + 2,0,0); + $terminal_exon->set_start($sc_start + 3); + $tx->add_feature($stop_codon); + } + } + } + } + } + if($found_start != -1){ + if(($phase eq ".") || + ($phase == 0)){ + if($sc[0] == 0){ + if($errors[31] < $max_error_count){ + print $warnings + "Unannotated start codon found on transcript ". + $tx->id."\n"; + } + $errors[31]++; + if (_check_errors_against_badlist($gtf,31)) { + push @bad_list, $tx->id; + } + if($fix_gtf){ + if($strand eq "+"){ + my $initial_exon = $$cds[0]; + my $sc_start = $initial_exon->start - 3 + $start; + my $start_codon = + GTF::Feature::new("start_codon",$sc_start, + $sc_start + 2,0,0); + $initial_exon->set_start($sc_start); + $tx->add_feature($start_codon); + } + else{ + my $initial_exon = $$cds[$#$cds]; + my $sc_start = $initial_exon->stop + 1 - $start; + my $start_codon = + GTF::Feature::new("start_codon",$sc_start, + $sc_start + 2,0,0); + $initial_exon->set_stop($sc_start + 2); + $tx->add_feature($start_codon); + } + } + } + } + } + # set phase if unset here; + if(($phase ne ".") && ($old_phase eq ".") && ($fix_gtf)){ + if($strand eq "+"){ + for(my $i = 0;$i <= $#$cds;$i++){ + $$cds[$i]->set_frame($phase); + $phase = ($$cds[$i]->length - $phase) % 3; + if($phase == 1){ + $phase = 2; + } + elsif($phase == 2){ + $phase = 1; + } + } + } + else{ + for(my $i = $#$cds;$i >= 0;$i--){ + $$cds[$i]->set_frame($phase); + $phase = ($$cds[$i]->length - $phase) % 3; + if($phase == 1){ + $phase = 2; + } + elsif($phase == 2){ + $phase = 1; + } + } + } + } + if($gtf->{Tx}){ + my $tx_fh = $gtf->{Tx}; + print $tx_fh ">".$tx->id ."\t".$tx->start."\t".$tx->stop."\n". + substr($seq,$start,$#seq - $stop - $start + 1)."\n"; + } + } + } + $gtf->{Total_Seq} = \%total_seq; + for(my $i = 0;$i <= $#error_msgs;$i++){ + if(($errors[$i] > 0) && !($$no_warn[$i])){ + print $warnings + "\nWarnings encountered:\n", + "Count\tDescription\n"; + for(my $j = $i;$j <= $#error_msgs;$j++){ + if(($errors[$j] > 0) && !($$no_warn[$j])){ + print $warnings + $errors[$j],"\t$error_msgs[$j]\n"; + } + } + last; + } + } + + #sort genes, transcripts, cds and utr by start position + @all_genes = sort {$a->start <=> $b->start} @all_genes; + @all_txs = sort {$a->start <=> $b->start} @all_txs; + @all_cds = sort {$a->start <=> $b->start} @all_cds; + @all_inter = sort {$a->start <=> $b->start} @all_inter; + @all_cns = sort {$a->start <=> $b->start} @all_cns; + + print $warnings "\nStatistics:\n"; + print $warnings + "\t", $#all_genes + 1, " genes with ", $#all_txs + 1, + " transcripts containing ", $#all_cds + 1 , " cds.\n"; + if($gtf->{Inframe_Stops}){ + print $warnings "Inframe Stop Genes:\n"; + foreach my $stop (@inframe_stops){ + print $warnings "$stop\n"; + } + } + if ($gtf->{Bad_List}) { + print $warnings "Bad Genes:\n"; + my %bad_out; + foreach my $bad (@bad_list) { + $bad_out{$bad}++; + } + my @out = keys %bad_out; + while (@out) { + print $warnings pop(@out),"\n"; + } + } + close(WARN); + close(GTF); + $gtf->{Genes} = \@all_genes; + $gtf->{Transcripts} = \@all_txs; + $gtf->{CDS} = \@all_cds; + $gtf->{Inter} = \@all_inter; + $gtf->{Inter_CNS} = \@all_cns; + $gtf->{Comments} = $all_line_comments; + $gtf->{Parse_Errors} = \@errors; +} + +# GTF::_get_error_messages() +# This function returns an array of the error messages associated with +# each error number. +sub _get_error_messages{ + my @error_msgs; + $error_msgs[0] = "Not enough fields on line."; + $error_msgs[1] = "Illegal type of whitespace between fields. Sould be tab."; + $error_msgs[2] = "Illegal value for field. Should be ". + "\'CDS\', \'exon\', \'start_codon\', or \'stop_codon\'."; + $error_msgs[3] = "Illegal value for field. Should be numerical."; + $error_msgs[4] = "Illegal value for field. Should be numerical."; + $error_msgs[5] = "Illegal value for field. Should be numerical or \'.\'."; + $error_msgs[6] = "Illegal value for field. ". + "Should be \'+\', \'-\', or \'.\'."; + $error_msgs[7] = "Illegal value for field. ". + "Should be \'0\', \'1\', \'2\', or \'.\'."; + $error_msgs[8] = "Illegal \'\#\' character in attribute name."; + $error_msgs[9] = "Illegal tab character in field."; + $error_msgs[10] = "Missing \';\' terminator after attribute in field."; + $error_msgs[11] = "Missing \'\"\'s around attribute values in field."; + $error_msgs[12] = "Incorrect separator between attribute name-value pairs in ". + " field. Should be \' \'."; + $error_msgs[13] = "Missing \'gene_id\' attribute in field."; + $error_msgs[14] = "Missing \'transcript_id\' attribute in field."; + $error_msgs[15] = "Transcript contains no CDS, inter or inter_CNS features."; + $error_msgs[16] = "Start Codon length is not three."; + $error_msgs[17] = "Transcript has no start codon."; + $error_msgs[18] = "Stop Codon length is not three."; + $error_msgs[19] = "Transcript has no stop codon."; + $error_msgs[20] = "Inconsistent value across gene_id."; + $error_msgs[21] = "CDS before start codon."; + $error_msgs[22] = "CDS after stop codon."; + $error_msgs[23] = "Multiple genes using the same transcript_id."; + $error_msgs[24] = "Bad start codon sequence. Should be ATG."; + $error_msgs[25] = "Bad stop codon sequence. Should be TAA, TAG, or TGA."; + $error_msgs[26] = "Bad acceptor splice site sequence."; + $error_msgs[27] = "Bad donor splice site sequence."; + $error_msgs[28] = "In frame stop codon found."; + $error_msgs[29] = "In frame stop codon in transcript."; + $error_msgs[30] = "Illegal \'=\' character in field."; + $error_msgs[31] = "Unannotated start codon found."; + $error_msgs[32] = "Unannotated stop codon found."; + $error_msgs[33] = "Start codon location inconsistent with initial exon location."; + $error_msgs[34] = "Stop codon location inconsistent with terminal exon location."; + $error_msgs[35] = "Start codon length is wrong."; + $error_msgs[36] = "Start codon length is wrong."; + $error_msgs[37] = "Transcript does not translate in any frame."; + $error_msgs[38] = " field is greater than field."; + $error_msgs[39] = "Wrong phase value."; + $error_msgs[40] = "Missing phase value."; + $error_msgs[41] = "Complete transcript length not multiple of 3."; + $error_msgs[42] = "Inconsistant phase value accross transcript."; + $error_msgs[43] = "Bad conservation sequence character. Should be \'0\',\'1\', ". + "or \'2\'."; + $error_msgs[44] = "Bad fasta sequence character. Should be \'A\',\'C\',\'T\',\'G\',". + "or \'N\'."; + $error_msgs[45] = "Overlapping CDS features in transcript."; + $error_msgs[46] = "Stop codon included in terminal CDS."; + $error_msgs[47] = "Start codon annotated in intron region."; + $error_msgs[48] = "Transcript contains zero length intron(s)."; + $error_msgs[49] = "Transcript contains short (<20bp) intron(s)."; + $error_msgs[50] = "Possible non-canonical splice site sequence."; + $error_msgs[51] = "Bad frame on SEC feature. Must be 0 or \".\""; + $error_msgs[52] = "SEC feature outside of CDS feature."; + $error_msgs[53] = "SEC feature had bad sequence. Must be \"TGA\""; + return @error_msgs; +} + +#this checks to see if the error is one of the errors that +#we want to return the transcript name and remove +sub _check_errors_against_badlist{ + my ($gtf,$error_num) = @_; + my $bad = $gtf->{Bad_List}; + if($$bad[$error_num]) { + return 1; + } + else { + return 0; + } +} + + +#this loads the gtf file without checking for errors +sub _load_no_check{ + my ($gtf,$warnings) = @_; + my (@all_genes,@all_txs,@all_cds,@all_inter,@all_cns); + my %TxObj; + my %GeneObj; + my $strict = $gtf->{Strict}; + #open file + open(GTF, "<$gtf->{Filename}") + or die "Could not open $gtf->{Filename}.\n"; + while(my $in_line = ){ + chomp $in_line; + if($in_line =~ /^(.*)\#/){ + $in_line = $1; + } + if($in_line =~ /^\s+(.*)$/){ + $in_line = $1; + } + if($in_line !~ /\S/){ + next; + } + my @data = split /\t/,$in_line; + if($#data != 8){ + die "Wrong number of fields on line:\n$in_line\n"; + } + if(!$strict && $data[5] eq "."){ + $data[5] = 0; + } + my %attributes; + while($data[8] =~ /^\s*(\S+) \"(\S*)\";(.*)$/){ + my $key = $1; + my $val = $2; + $data[8] = $3; + $attributes{$key} = $val; + } + my $gid; + if(defined($attributes{gene_id})){ + $gid = $attributes{gene_id}; + } + else{ + die "Missing gene_id\n"; + } + my $tid; + if(defined($attributes{transcript_id})){ + $tid = $attributes{transcript_id}; + } + else{ + die "Missing transcript_id\n"; + } + #create objects + my $feature = + GTF::Feature::new($data[2], $data[3], $data[4], $data[5], $data[7]); + if($feature->type eq "CDS"){ + push @all_cds, $feature; + } + elsif($feature->type eq "inter") { + push @all_inter, $feature; + } + elsif($feature->type eq "inter_CNS") { + push @all_cns, $feature; + } + + my $tx; + my $gene; + unless(defined($TxObj{$tid}) && length($tid)){ + $tx = GTF::Transcript::new($tid); + $TxObj{$tid} = $tx; + unless($feature->type eq "inter" || $feature->type eq "inter_CNS") { + push @all_txs, $tx; + } + unless(defined($GeneObj{$gid}) && length($gid)){ + $gene = GTF::Gene::new($gid,$data[0],$data[1],$data[6]); + $GeneObj{$gid} = $gene; + unless($feature->type eq "inter" || $feature->type eq "inter_CNS") { + push @all_genes, $gene; + } + } + $gene = $GeneObj{$gid}; + $gene->add_transcript($tx); + } + else{ + $gene = $GeneObj{$gid}; + } + $tx = $TxObj{$tid}; + if($tx->strand ne $data[6]){ + die "Features on transcript ".$tx->id." have different strands\n"; + } + $tx->add_feature($feature); + } + close(GTF); + #sort genes, exons, and cds by start position + @all_genes = sort {$a->start <=> $b->start} @all_genes; + @all_txs = sort {$a->start <=> $b->start} @all_txs; + @all_cds = sort {$a->start <=> $b->start} @all_cds; + @all_inter = sort {$a->start <=> $b->start} @all_inter; + @all_cns = sort {$a->start <=> $b->start} @all_cns; + $gtf->{Genes} = \@all_genes; + $gtf->{Transcripts} = \@all_txs; + $gtf->{CDS} = \@all_cds; + $gtf->{Inter_CNS} = \@all_cns; + $gtf->{Inter} = \@all_inter; +} + +sub _rev_comp{ + my ($seq) = @_; + $seq = uc $seq; + my @bases = split //, $seq; + my $out = ""; + foreach my $base (@bases){ + if($base eq 'A'){ + $base = 'T'; + } + elsif($base eq 'C'){ + $base = 'G'; + } + elsif($base eq 'G'){ + $base = 'C'; + } + elsif($base eq 'T'){ + $base = 'A'; + } + $out = $base.$out; + } + return $out; + +} + +############################################################################### +# GTF::Gene +############################################################################### +# The GTF::Gene object stores all gtf data for one gene. A gene contains one +# or more transcripts +package GTF::Gene; +# GTF::Gene::new(gene_id,seqname,source,strand) +# This is the contrcutor for a GTF::Gene object. It has 4 required +# parameters. The are: +# gene_id - The gene_id for this gene +# seqname - The field of the line. Should be a string +# containing no whitespace. +# source - The field of the line. Should be a string +# containing no whitespace. +# strand - The field of the line. Should be either +# '+', '-', or '.'. +sub new { + my $gene = bless {}; + ($gene->{Id},$gene->{Seqname}, $gene->{Source}, $gene->{Strand}) = @_; + $gene->{Transcripts} = []; + $gene->{CDS} = []; + $gene->{Modified} = 0; + $gene->{Start} = -1; + $gene->{Stop} = -1; + return $gene; +} + +sub add_transcript{ + my ($gene,$tx) = @_; + my $txs = $gene->{Transcripts}; + push @$txs, $tx; + $tx->_set_gene($gene); + $tx->_update; + $gene->{Modified} = 1; +} + +sub id {shift->{Id}} + +sub set_id{ + my ($gene,$id) = @_; + $gene->{Id} = $id; +} + +# GTF::Feature::seqname() +# This function returns the sequnce name, field, that this +# feature came from as a string. +sub seqname {shift->{Seqname}} + +# GTF::Feature::set_source() +# This function sets the sequence name, field, of this line +sub set_seqname{ + my ($feature,$seqname) = @_; + $feature->{Seqname} = $seqname; +} + +# GTF::Feature::source() +# This function returns the source, field, of this line as a +# string. +sub source {shift->{Source}} + +# GTF::Feature::set_source() +# This function sets the source, field, of this line +sub set_source{ + my ($feature,$source) = @_; + $feature->{Source} = $source; +} + +# GTF::Feature::strand() +# This function returns the strand, field, of this line as a +# string. Should be '+', '-', or '.'. +sub strand {shift->{Strand}} + +# GTF::Gene::id() +# This function returns this transcript's unique transcript_id +sub gene_id{ + my ($gene) = @_; + return $gene->{Id}; +} + +# GTF::Gene::transcript_ids() +# This function returns an array of all trnacsripts in this gene +sub transcripts{ + my ($gene) = @_; + $gene->_update; + return $gene->{Transcripts};# [sort {$a->start <=> $b->start || $a->stop <=> $b->stop} @{$gene->{Transcripts}}]; +} + +# GTF::Gene::remove_transcript() +sub remove_transcript{ + my($gene, $tx_id) = @_; + my $txs = $gene->transcripts; + my @new_txs = (); + + @new_txs = grep($_->id ne $tx_id, @$txs); + $gene->{Transcripts} = \@new_txs if ($#new_txs < $#$txs); +} + +sub cds{ + my ($gene) = @_; + + my $txs = $gene->transcripts; + my @cds; + foreach my $tx (@$txs){ + push @cds, @{$tx->cds}; + } + + # employ a slow sort to avoid perl sort problems (see GTF::Gene::_update) rpz + my $tmp; + for(my $a = 0; $a <= $#cds; $a++){ + for(my $b = $a+1; $b <= $#cds; $b++){ + if(($cds[$a]->start > $cds[$b]->start) || + (($cds[$a]->start == $cds[$b]->start) && + ($cds[$a]->stop > $cds[$b]->stop))){ + $tmp = $cds[$a]; + $cds[$a] = $cds[$b]; + $cds[$b] = $tmp; + } + } + } + + return \@cds; +} + +# GTF::Gene::copy() +# This function returns a new object which is a copy of this object. Also +# creates copies of all transcripts of this gene +sub copy{ + my ($gene) = @_; + my $copy = GTF::Gene::new($gene->id,$gene->seqname,$gene->source,$gene->strand); + my $txs = $gene->transcripts; + my @new_txs; + my $next = 0; + foreach my $tx (@$txs){ + $next++; + my $ntx = $tx->copy; + $copy->add_transcript($ntx); + } + return($copy); +} + +# GTF::Gene::offset(ammount) +# This function will offset all locations in this gene by the given ammount +sub offset{ + my ($gene,$offset) = @_; + my $txs = $gene->transcripts; + foreach my $tx (@$txs){ + $tx->offset($offset); + } +} + +# GTF::Gene::output_gtf([file_handle]) +# This function outputs the information for this gene in standard +# gtf2 format. If takes and optional file_handle as the only argument +# to which it outputs the data. If no argument is given it outputs +# the data to stdout. +sub output_gtf{ + my ($gene,$out_handle) = @_; + unless(defined($out_handle)){ + $out_handle = \*STDOUT; + } + my $txs = $gene->transcripts; + foreach my $tx (@$txs){ + $tx->output_gtf($out_handle); + print $out_handle "\n"; + } +} + +# GTF::Gene::output_gff([file_handle]) +# This function outputs the information for this gene in standard +# GFF format. If takes and optional file_handle as the only argument +# to which it outputs the data. If no argument is given it outputs +# the data to stdout. +sub output_gff{ + my ($gene,$out_handle) = @_; + unless(defined($out_handle)){ + $out_handle = \*STDOUT; + } + my $txs = $gene->transcripts; + foreach my $tx (@$txs){ + $tx->output_gff($out_handle); + } +} + +# GTF::Gene::reverse_complement(seq_length) +# This function takes the length of the sequence this gene came from and +# reverse complements everything in the gene. +sub reverse_complement{ + my ($gene, $seq_length) = @_; + my $txs = $gene->transcripts; + foreach my $tx (@$txs){ + $tx->reverse_complement($seq_length); + } + $gene->{Modified} = 1; +} + +# GTF::Gene::start() +# This function return the lowest coord of any feature in any tx of this gene +sub start{ + my ($gene) = @_; + $gene->_update; + return $gene->{Start}; +} + +# GTF::Gene::stop() +# This function returns the highest coord of any feature in any tx of this gene +sub stop{ + my ($gene) = @_; + $gene->_update; + return $gene->{Stop}; +} + +sub equals{ + my ($gene,$compare) = @_; + $gene->_update; + unless($gene && $compare){ + return 0; + } + unless(($gene->start == $compare->start) && + ($gene->stop == $compare->stop) && + ($gene->strand eq $compare->strand)){ + return 0; + } + my $gene_txs = $gene->transcripts; + my $compare_txs = $compare->transcripts; + unless($#$gene_txs == $#$compare_txs){ + return 0; + } + for(my $i = 0;$i <= $#$gene_txs;$i++){ + unless($$gene_txs[$i]->equals($$compare_txs[$i])){ + return 0; + } + } + return 1; +} + +sub length{ + my ($gene) = @_; + my $txs = $gene->transcripts; + if($#$txs == -1){ + return 0; + } + my $start = $$txs[0]->start; + my $stop = $$txs[0]->stop; + my $total = 0; + foreach my $tx (@$txs){ + if($tx->start < $start){ + $start = $tx->start; + } + if($tx->stop > $stop){ + $stop = $tx->stop; + } + } + return($stop - $start + 1); +} + +sub gc_percentage{ + my ($gene) = @_; + my $txs = $gene->transcripts; + if($#$txs == -1){ + return 0; + } + my $total = 0; + foreach my $tx (@$txs){ + $total += $tx->gc_percentage; + } + return($total/($#$txs+1)); +} + +sub match_percentage{ + my ($gene) = @_; + my $txs = $gene->transcripts; + if($#$txs == -1){ + return 0; + } + my $total = 0; + foreach my $tx (@$txs){ + $total += $tx->match_percentage; + } + return($total/($#$txs+1)); +} + +sub mismatch_percentage{ + my ($gene) = @_; + my $txs = $gene->transcripts; + if($#$txs == -1){ + return 0; + } + my $total = 0; + foreach my $tx (@$txs){ + $total += $tx->mismatch_percentage; + } + return($total/($#$txs+1)); +} + +sub unaligned_percentage{ + my ($gene) = @_; + my $txs = $gene->transcripts; + if($#$txs == -1){ + return 0; + } + my $total = 0; + foreach my $tx (@$txs){ + $total += $tx->unaligned_percentage; + } + return($total/($#$txs+1)); +} + +sub tag{ + my ($gene) = @_; + return $gene->{Tag}; +} + +sub set_tag{ + my ($gene,$tag) = @_; + $gene->{Tag} = $tag; +} + +sub _update{ + my ($gene) = @_; + unless($gene->{Modified}){ + return; + } + $gene->{Modified} = 0; + my $txs = $gene->{Transcripts}; + + + # this slow sort is being done because the commented out sort below + # causes crashes sometimes and I have no idea why since the uncommented + # sort has no problems + # -- the reason for the problems is that perl's sort routine does not + # afford you the capability of modifying the $a or $b variable in + # the usersub callback, since they are aliases. rpz + $gene->{Start} = -1; + $gene->{Stop} = -1; + my $tmp; + for(my $a = 0; $a <= $#$txs; $a++){ + for(my $b = $a+1; $b <= $#$txs; $b++){ + if(($$txs[$a]->start > $$txs[$b]->start) || + (($$txs[$a]->start == $$txs[$b]->start) && + ($$txs[$a]->stop > $$txs[$b]->stop))){ + $tmp = $$txs[$a]; + $$txs[$a] = $$txs[$b]; + $$txs[$b] = $tmp; + } + } + if($$txs[$a]->stop > $gene->{Stop}){ + # highest stop may not be stop of last tx so we check them all + $gene->{Stop} = $$txs[$a]->stop; + } + } + if($#$txs != -1){ + # lowest start is start of first tx + $gene->{Start} = $$txs[0]->start; + } +} + +############################################################################### +# GTF::Transcript +############################################################################### +# The GTF::Transcript object stores all data for a single transcript of a gtf +# file. +package GTF::Transcript; +# GTF::Transcript::new(transcript_id) +# This is the constructor for GTF::Feature objects. It takes 1 arguments. +# transciprt_id - the id for this transcript +sub new{ + my $tx = bless {}; + ($tx->{Id}) = @_; + $tx->{Exons} = []; + $tx->{CDS} = []; + $tx->{UTR3} = []; + $tx->{UTR5} = []; + $tx->{Introns} = []; + $tx->{Starts} = []; + $tx->{Stops} = []; + $tx->{SEC} = []; + $tx->{Intron_CNS} = []; + $tx->{Inter_CNS} = []; + $tx->{Inter} = []; + $tx->{Coding_Start} = -1; + $tx->{Coding_Stop} = -1; + $tx->{Coding_Length} = -1; + $tx->{Gene} = -1; + $tx->{Start} = -1; + $tx->{Stop} = -1; + $tx->{Modified} = 1; + return $tx; +} + +sub _set_gene{ + my ($tx,$gene) = @_; + $tx->{Gene} = $gene; +} + +sub add_feature{ + my ($tx,$feature) = @_; + my $type = $feature->type; + if($type eq 'CDS'){ + push @{$tx->{CDS}}, $feature; + $tx->{CDS} = [sort {$a->start <=> $b->start} @{$tx->{CDS}}]; + } + elsif($type eq '3UTR'){ + push @{$tx->{UTR3}}, $feature; + $tx->{UTR3} = [sort {$a->start <=> $b->start} @{$tx->{UTR3}}]; + } + elsif($type eq '5UTR'){ + push @{$tx->{UTR5}}, $feature; + $tx->{UTR5} = [sort {$a->start <=> $b->start} @{$tx->{UTR5}}]; + } + elsif($type eq 'start_codon'){ + push @{$tx->{Starts}}, $feature; + $tx->{Starts} = [sort {$a->start <=> $b->start} @{$tx->{Starts}}]; + } + elsif($type eq 'stop_codon'){ + push @{$tx->{Stops}}, $feature; + $tx->{Stops} = [sort {$a->start <=> $b->start} @{$tx->{Stops}}]; + } + elsif($type eq 'exon'){ + push @{$tx->{Exons}}, $feature; + $tx->{Exons} = [sort {$a->start <=> $b->start} @{$tx->{Exons}}]; + } + elsif($type eq 'SEC'){ + push @{$tx->{SEC}}, $feature; + $tx->{SEC} = [sort {$a->start <=> $b->start} @{$tx->{SEC}}]; + } + elsif($type eq 'intron_CNS'){ + push @{$tx->{Intron_CNS}}, $feature; + $tx->{Intron_CNS} = [sort {$a->start <=> $b->start} @{$tx->{Intron_CNS}}]; + } + elsif($type eq 'inter_CNS'){ + push @{$tx->{Inter_CNS}}, $feature; + $tx->{Inter_CNS} = [sort {$a->start <=> $b->start} @{$tx->{Inter_CNS}}]; + } + elsif($type eq 'inter'){ + push @{$tx->{Inter}}, $feature; + $tx->{Inter} = [sort {$a->start <=> $b->start} @{$tx->{Inter}}]; + } + $feature->set_transcript($tx); + $tx->{Modified} = 1; + if($tx->{Gene} != -1){ $tx->{Gene}->{Modified} = 1;} +} + +sub transfer_missing_utr{ + my($dest, $src, $mode) = @_; + my $dest_utr3 = $dest->utr3; + my $dest_utr5 = $dest->utr5; + my $src_utr3 = $src->utr3; + my $src_utr5 = $src->utr5; + + $mode = "Full" if ($#_ == 1); # default + + if (($mode eq "3UTR") || ($mode eq "Full")) { + if (($#$dest_utr3 == -1) && ($#$src_utr3 > -1)){ + print "Transfering 3UTR from ".$src->id." to ".$dest->id."\n"; + foreach my $src_exon (@$src_utr3) + { + my $utr_exon = $src_exon->copy; + $utr_exon->set_transcript($dest); + $dest->add_feature($utr_exon); + } + } + } + if (($mode eq "5UTR") || ($mode eq "Full")) { + if (($#$dest_utr5 == -1) && ($#$src_utr5 > -1)){ + print "Transfering 5UTR from ".$src->id." to ".$dest->id."\n"; + foreach my $src_exon (@$src_utr5) + { + my $utr_exon = $src_exon->copy; + $utr_exon->set_transcript($dest); + $dest->add_feature($utr_exon); + } + } + } +} + +# copies the exon obejcts info into utr5 and utr3 objects +# WARNING: don't call this if you already have utr5 and utr3 info +sub create_utr_objects_from_exons{ + my ($tx) = @_; + my $exons = $tx->exons; + my $start = $tx->coding_start; + my $stop = $tx->coding_stop; + if($start != -1){ + my $type = "5UTR"; + if($tx->strand eq "-"){ + $type = "3UTR"; + } + for(my $i = 0; $i <= $#$exons; $i++){ + if($$exons[$i]->start < $start){ + my $utr = $$exons[$i]->copy; + $utr->set_type($type); + if($start < $utr->stop){ + $utr->set_stop($start); + } + $tx->add_feature($utr); + } + else{ + last; + } + } + } + if($stop != -1){ + my $type = "3UTR"; + if($tx->strand eq "-"){ + $type = "5UTR"; + } + for(my $i = $#$exons; $i >= 0; $i--){ + if($$exons[$i]->stop > $stop){ + my $utr = $$exons[$i]->copy; + $utr->set_type($type); + if($stop > $utr->start){ + $utr->set_start($stop); + } + $tx->add_feature($utr); + } + else{ + last; + } + } + } +} + +sub exons{ + my ($tx) = @_; + $tx->_update; + return $tx->{Exons}; +} + +sub cds{ + my ($tx) = @_; + $tx->_update; + return $tx->{CDS}; +} + +sub utr3{ + my ($tx) = @_; + $tx->_update; + return $tx->{UTR3}; +} + +sub utr5{ + my ($tx) = @_; + $tx->_update; + return $tx->{UTR5}; +} + +sub introns{ + my ($tx) = @_; + $tx->_update; + return $tx->{Introns}; +} + +sub start_codons{ + my ($tx) = @_; + $tx->_update; + return $tx->{Starts}; +} + +sub stop_codons{ + my ($tx) = @_; + $tx->_update; + return $tx->{Stops}; +} + +sub sec{ + my ($tx) = @_; + $tx->_update; + return $tx->{SEC}; +} + +sub intron_cns{ + my ($tx) = @_; + $tx->_update; + return $tx->{Intron_CNS}; +} + +sub inter_cns{ + my ($tx) = @_; + $tx->_update; + return $tx->{Inter_CNS}; +} + +sub inter{ + my ($tx) = @_; + $tx->_update; + return $tx->{Inter}; +} + +sub seqname{ + my ($tx) = @_; + my $gene = $tx->gene; + return $gene->seqname; +} + +sub source{ + my ($tx) = @_; + my $gene = $tx->gene; + return $gene->source; +} + +sub strand{ + my ($tx) = @_; + my $gene = $tx->gene; + return $gene->strand; +} + +sub set_id{ + my ($tx,$id) = @_; + $tx->{Id} = $id; +} + +sub id{ + my ($tx) = @_; + return $tx->{Id}; +} + +sub gene_id{ + my ($tx) = @_; + my $gene = $tx->gene; + return $gene->{Id}; +} + +sub gene{ + my ($tx) = @_; + return $tx->{Gene}; +} + +# GTF::Transcript::remove_exon(exon start pos) +# Allows you to remove an exon from the list +# Removes the all exons at the specified position +sub remove_exon{ + my ($tx, $pos) = @_; + my $new_exons = []; + my $removed = 0; + foreach my $exon (@{$tx->{Exons}}){ + if($exon->start != $pos){ + push @$new_exons, $exon; + } + else{ + $removed++; + } + } + $tx->{Exons} = $new_exons; + return $removed; +} + +# GTF::Transcript::remove_cds(cds start pos) +# Allows you to remove a cds from the list +# Removes the all cdss at the specified position +sub remove_cds{ + my ($tx, $pos) = @_; + my $new_cds = []; + my $removed = 0; + foreach my $cds (@{$tx->{CDS}}){ + if($cds->start != $pos){ + push @$new_cds, $cds; + } + else{ + $removed++; + } + } + $tx->{CDS} = $new_cds; + return $removed; +} + +# GTF::Transcript::length() +# This function return the length of the tx from the start to the stop. +sub length{ + my ($tx) = @_; + return($tx->stop - $tx->start + 1); +} + +# GTF::Transcript::start() +# This function returns the start of this tx (lowest coord). +sub start{ + my ($tx) = @_; + $tx->_update; + return $tx->{Start}; +} + +# GTF::Transcript::stop() +# This function return the end of the tx (highest coord). +sub stop{ + my ($tx) = @_; + $tx->_update; + return $tx->{Stop}; +} + +# GTF::Transcript::coding_start() +# This function returns the start position (lowest coord) of coding region of +# this tx. Same as the start codon's field. +sub coding_start{ + my ($tx) = @_; + $tx->_update; + return $tx->{Coding_Start}; +} + +# GTF::Transcript::coding_stop() +# This function return the end position (highest coord) of the coding region of +# the tx. Same as the stop codon's field. +sub coding_stop{ + my ($tx) = @_; + $tx->_update; + return $tx->{Coding_Stop}; +} + +# GTF::Transcript::coding_length() +# This function return the coding_length of the tx from the coding_start +# to the coding_stop. +sub coding_length{ + my ($tx) = @_; + $tx->_update; + return $tx->{Coding_Length}; +} + +sub score{ + my ($tx) = @_; + my $features = $tx->_all_features; + my $score = 0; + foreach my $feature (@$features){ + if($feature->score ne "."){ + $score += $feature->score; + } + } + return $score; +} + +# GTF::Transcript::initial_exon() +# This function returns a CDS::Feature object representing the +# initial_exon (cds) of this tx +# NOTE: Isn't exon the wrong word for this? +sub initial_exon{ + my ($tx) = @_; + my $starts = $tx->start_codons; + if($#$starts == -1){ + return 0; + } + $tx->_update; + my $exons = $tx->{CDS}; + if($tx->strand eq "-"){ + return $$exons[$#{$exons}]; + } + else{ + return $$exons[0]; + } +} + +# GTF::Transcript::terminal_exon() +# This function returns a CDS::Feature object representing the +# terminal exon (cds) of this tx +# NOTE: Isn't exon the wrong word for this? +sub terminal_exon{ + my ($tx) = @_; + my $stops = $tx->stop_codons; + if($#$stops == -1){ + return 0; + } + $tx->_update; + my $exons = $tx->cds; + if($tx->strand eq "-"){ + return $$exons[0]; + } + else{ + return $$exons[$#{$exons}]; + } +} + +sub equals{ + my ($tx,$compare, $mode) = @_; + $mode = "Full" if ($#_ == 1); # default + + unless($tx && $compare){ + return 0; + } + unless(($tx->start == $compare->start) && + ($tx->stop == $compare->stop) && + ($tx->strand eq $compare->strand)){ + return 0; + } + + if (($mode eq "CDS") || ($mode eq "Full")) { + my $tx_cds = $tx->cds; + my $comp_cds = $compare->cds; + unless($#$tx_cds == $#$comp_cds){ + return 0; + } + for(my $i = 0;$i <= $#$tx_cds;$i++){ + unless($$tx_cds[$i]->equals($$comp_cds[$i])){ + return 0; + } + } + my $tx_starts = $tx->start_codons; + my $comp_starts = $compare->start_codons; + unless($#$tx_starts == $#$comp_starts){ + return 0; + } + for(my $i = 0;$i <= $#$tx_starts;$i++){ + unless($$tx_starts[$i]->equals($$comp_starts[$i])){ + return 0; + } + } + my $tx_stops = $tx->stop_codons; + my $comp_stops = $compare->stop_codons; + unless($#$tx_stops == $#$comp_stops){ + return 0; + } + for(my $i = 0;$i <= $#$tx_stops;$i++){ + unless($$tx_stops[$i]->equals($$comp_stops[$i])){ + return 0; + } + } + } + + if (($mode eq "3UTR") || ($mode eq "Full")) { + my $tx_utr3 = $tx->utr3; + my $comp_utr3 = $compare->utr3; + unless($#$tx_utr3 == $#$comp_utr3){ + return 0; + } + for(my $i = 0;$i <= $#$tx_utr3;$i++){ + unless($$tx_utr3[$i]->equals($$comp_utr3[$i])){ + return 0; + } + } + } + + if (($mode eq "5UTR") || ($mode eq "Full")) { + my $tx_utr5 = $tx->utr5; + my $comp_utr5 = $compare->utr5; + unless($#$tx_utr5 == $#$comp_utr5){ + return 0; + } + for(my $i = 0;$i <= $#$tx_utr5;$i++){ + unless($$tx_utr5[$i]->equals($$comp_utr5[$i])){ + return 0; + } + } + } + + my $tx_exon = $tx->exons; + my $comp_exon = $compare->exons; + unless($#$tx_exon == $#$comp_exon){ + return 0; + } + for(my $i = 0;$i <= $#$tx_exon;$i++){ + unless($$tx_exon[$i]->equals($$comp_exon[$i])){ + return 0; + } + } + + my $tx_sec = $tx->sec; + my $comp_sec = $compare->sec; + unless($#$tx_sec == $#$comp_sec){ + return 0; + } + for(my $i = 0;$i <= $#$tx_sec;$i++){ + unless($$tx_sec[$i]->equals($$comp_sec[$i])){ + return 0; + } + } + return 1; +} + +# GTF::Transcript::offset(ammount) +# This function will offset all locations in this tx by the given ammount +sub offset{ + my ($tx,$offset) = @_; + unless(defined($offset)){ + print STDERR "Undefined value passed to GTF::Gene::offset.\n"; + return; + } + my $features = $tx->_all_features; + foreach my $feature (@$features){ + if(defined($feature)){ + $feature->offset($offset); + } + } + $tx->{Modified} = 1; + if($tx->{Gene} != -1){ $tx->{Gene}->{Modified} = 1;} +} + +# GTF::Transcript::output_gtf([file_handle]) +# This function outputs the information for this tx in standard +# gtf2 format. If takes and optional file_handle as the only argument +# to which it outputs the data. If no argument is given it outputs +# the data to stdout. +sub output_gtf{ + my ($tx,$out_handle) = @_; + unless(defined $out_handle){ + $out_handle = \*STDOUT; + } + my @features = + sort {$a->start <=> $b->start || $a->stop <=> $b->stop} @{$tx->_all_features}; + foreach my $feature (@features){ + $feature->output_gtf($out_handle); + } +} + +# GTF::Transcript::output_gff([file_handle]) +# This function outputs the information for this tx in standard +# GFF format. If takes and optional file_handle as the only argument +# to which it outputs the data. If no argument is given it outputs +# the data to stdout. +sub output_gff{ + my ($tx,$out_handle) = @_; + unless(defined $out_handle){ + $out_handle = \*STDOUT; + } + my @features = + sort {$a->start <=> $b->start || $a->stop <=> $b->stop} @{$tx->_all_features}; + foreach my $feature (@features){ + $feature->output_gff($out_handle); + } +} + +# GTF::Transcript::reverse_complement(seq_length) +# This function takes the length of the sequence this tx came from and +# reverse complements everything in the gene. +sub reverse_complement{ + my ($tx, $seq_length) = @_; + unless(defined($seq_length)){ + print STDERR + "Undefined value for seq_length passed to GTF::Gene::reverse_complement.\n"; + } + my $features = $tx->_all_features; + foreach my $feature (@$features){ + $feature->reverse_complement($seq_length); + } +} + +sub _all_features{ + my ($tx) = @_; + my @features; + my $starts = $tx->start_codons; + push @features, @$starts; + my $stops = $tx->stop_codons; + push @features, @$stops; + my $exons = $tx->exons; + push @features, @$exons; + my $cds = $tx->cds; + push @features, @$cds; + my $utr3 = $tx->utr3; + push @features, @$utr3; + my $utr5 = $tx->utr5; + push @features, @$utr5; + my $sec = $tx->sec; + push @features, @$sec; + my $cns = $tx->intron_cns; + push @features, @$cns; + return \@features; +} + +sub copy{ + my ($tx) = @_; + my $copy = GTF::Transcript::new($tx->id); + my $starts = $tx->start_codons; + foreach my $old (@$starts){ + my $new = $old->copy; + $copy->add_feature($new); + } + my $stops = $tx->stop_codons; + foreach my $old (@$stops){ + my $new = $old->copy; + $copy->add_feature($new); + } + my $exons = $tx->exons; + foreach my $old (@$exons){ + my $new = $old->copy; + $copy->add_feature($new); + } + my $cds = $tx->cds; + foreach my $old (@$cds){ + my $new = $old->copy; + $copy->add_feature($new); + } + my $utr3 = $tx->utr3; + foreach my $old (@$utr3){ + my $new = $old->copy; + $copy->add_feature($new); + } + my $utr5 = $tx->utr5; + foreach my $old (@$utr5){ + my $new = $old->copy; + $copy->add_feature($new); + } + my $sec = $tx->sec; + foreach my $old (@$sec){ + my $new = $old->copy; + $copy->add_feature($new); + } + return($copy); +} + +sub gc_percentage{ + my ($tx) = @_; + my $features = $tx->_all_features; + my $total = 0; + my $percent = 0; + foreach my $feature (@$features){ + $percent += $feature->length * $feature->gc_percentage; + $total += $feature->length; + } + $percent /= $total; + return $percent; +} + +sub match_percentage{ + my ($tx) = @_; + my $features = $tx->_all_features; + my $total = 0; + my $percent = 0; + foreach my $feature (@$features){ + $percent += $feature->length * $feature->match_percentage; + $total += $feature->length; + } + $percent /= $total; + return $percent; +} + +sub mismatch_percentage{ + my ($tx) = @_; + my $features = $tx->_all_features; + my $total = 0; + my $percent = 0; + foreach my $feature (@$features){ + $percent += $feature->length * $feature->mismatch_percentage; + $total += $feature->length; + } + $percent /= $total; + return $percent; +} + +sub unaligned_percentage{ + my ($tx) = @_; + my $features = $tx->_all_features; + my $total = 0; + my $percent = 0; + foreach my $feature (@$features){ + $percent += $feature->length * $feature->unaligned_percentage; + $total += $feature->length; + } + $percent /= $total; + return $percent; +} + +sub tag{ + my ($tx) = @_; + return $tx->{Tag}; +} + +sub set_tag{ + my ($tx,$tag) = @_; + $tx->{Tag} = $tag; +} + +sub _update{ + my ($tx) = @_; + unless($tx->{Modified}){ + return; + } + $tx->{Modified} = 0; + my $starts = $tx->start_codons; + my $stops = $tx->stop_codons; + my $cds = $tx->cds; + my $utr5 = $tx->utr5; + my $utr3 = $tx->utr3; + my $exons = $tx->exons; + foreach my $utr (@$utr5){ + $utr->set_subtype("UTR5"); + } + foreach my $utr (@$utr3){ + $utr->set_subtype("UTR3"); + } + #set subtypes + if($#$cds == 0){ + if(($#$starts >= 0) && ($#$stops >= 0)){ + $$cds[0]->set_subtype("Single"); + } + elsif($#$starts >= 0){ + $$cds[0]->set_subtype("Initial"); + } + elsif($#$stops >= 0){ + $$cds[0]->set_subtype("Terminal"); + } + else{ + $$cds[0]->set_subtype("Internal"); + } + } + elsif($#$cds >= 0){ + my $strand = $tx->strand; + my $starts = $tx->start_codons; + my $stops = $tx->stop_codons; + my $start = 0; + my $stop = $#$cds; + if($strand eq '-'){ + if($#$starts >= 0){ + $$cds[$#$cds]->set_subtype("Initial"); + $stop--; + } + if($#$stops >= 0){ + $$cds[0]->set_subtype("Terminal"); + $start++; + } + } + else{ + if($#$starts >= 0){ + $$cds[0]->set_subtype("Initial"); + $start++; + } + if($#$stops >= 0){ + $$cds[$#$cds]->set_subtype("Terminal"); + $stop--; + } + } + for(my $i = $start;$i <= $stop;$i++){ + $$cds[$i]->set_subtype("Internal"); + } + } + #get introns and coding length + my $clen = 0; + my @introns; + for(my $i = 0;$i < $#$cds;$i++){ + $clen += $$cds[$i]->stop - $$cds[$i]->start +1; + my $start = $$cds[$i]->stop + 1; + my $stop = $$cds[$i+1]->start - 1; + if($start <= $stop){ + my $intron = GTF::Feature::new("Intron",$start,$stop,0,'.'); + $intron->set_subtype("Intron"); + $intron->set_transcript($tx); + push @introns, $intron; + } + } + if($#$cds >= 0){ + $clen += $$cds[$#$cds]->stop - $$cds[$#$cds]->start +1; + } + $tx->{Coding_Length} = $clen; + $tx->{Introns} = \@introns; + #get start/coding start + my $start = -1; + if(($#{$cds} >= 0) && + (($start == -1) || ($start > $$cds[0]->start))){ + $start = $$cds[0]->start; + } + $tx->{Coding_Start} = $start; + if(($#{$starts} >= 0) && + (($start == -1) || ($start > $$starts[0]->start))){ + $start = $$starts[0]->start; + } + if(($#{$stops} >= 0) && + (($start == -1) || ($start > $$stops[0]->start))){ + $start = $$stops[0]->start; + } + if(($#{$exons} >= 0) && + (($start == -1) || ($start > $$exons[0]->start))){ + $start = $$exons[0]->start; + } + if(($#{$utr5} >= 0) && + (($start == -1) || ($start > $$utr5[0]->start))){ + $start = $$utr5[0]->start; + } + if(($#{$utr3} >= 0) && + (($start == -1) || ($start > $$utr3[0]->start))){ + $start = $$utr3[0]->start; + } + $tx->{Start} = $start; + #get stop/coding stop + my $stop = -1; + if(($#{$cds} >= 0) && + (($stop == -1) || ($stop < $$cds[$#$cds]->stop))){ + $stop = $$cds[$#$cds]->stop; + } + $tx->{Coding_Stop} = $stop; + if(($#{$starts} >= 0) && + (($stop == -1) || ($stop < $$starts[$#$starts]->stop))){ + $stop = $$starts[$#$starts]->stop; + } + if(($#{$stops} >= 0) && + (($stop == -1) || ($stop < $$stops[$#$stops]->stop))){ + $stop = $$stops[$#$stops]->stop; + } + if(($#{$exons} >= 0) && + (($stop == -1) || ($stop < $$exons[$#$exons]->stop))){ + $stop = $$exons[$#$exons]->stop; + } + if(($#{$utr5} >= 0) && + (($stop == -1) || ($stop < $$utr5[$#$utr5]->stop))){ + $stop = $$utr5[$#$utr5]->stop; + } + if(($#{$utr3} >= 0) && + (($stop == -1) || ($stop < $$utr3[$#$utr3]->stop))){ + $stop = $$utr3[$#$utr3]->stop; + } + $tx->{Stop} = $stop; +} + + + +############################################################################### +# GTF::Feature +############################################################################### +# The GTF::Feature object stores all data for a single feature of a gtf +# file. This object basically stores all data for a single non-comment +# line of a gtf file. +package GTF::Feature; +# GTF::Feature::new(feature, start, end, score, frame) +# This is the constructor for GTF::Feature objects. It takes 5 required +# arguments. They are: +# type - The field of the line. Should be either +# 'cds', 'exon', 'start_codon', or 'stop_codon'. +# start - The field of the line. Should be a number. +# end - The field of the line. Should be a number. +# score - The field of the line. Should be a number or '.'. +# frame - The field of the line. Should be either +# '0', '1', '2', or '.'. +sub new { + my $feature = bless {}; + ($feature->{Type}, $feature->{Start}, $feature->{End}, $feature->{Score}, + $feature->{Frame}) = @_; + my $start = $feature->{Start}; + my $stop = $feature->{Stop}; + if(defined($start) && defined($stop) && ($start > $stop)){ + $feature->{Start} = $stop; + $feature->{Stop} = $start; + } + $feature->{Transcript} = 0; + $feature->{Seq} = 0; + $feature->{Conseq} = 0; + $feature->{Subtype} = 0; + $feature->{ASE} = 0; # rpz - Alternative Splicing Event + $feature->{Match} = -1; + $feature->{Mismatch} = -1; + $feature->{Unaligned} = -1; + $feature->{GC} = -1; + $feature->{A} = -1; + $feature->{C} = -1; + $feature->{G} = -1; + $feature->{T} = -1; + $feature->{N} = -1; + $feature->{0} = -1; + $feature->{1} = -1; + $feature->{2} = -1; + return $feature; +} + +# GTF::Feature::copy() +# This function returns a new object which is a copy of this object +sub copy{ + my ($feature) = @_; + my $copy = GTF::Feature::new($feature->type,$feature->start,$feature->stop, + $feature->score,$feature->frame); + $copy->set_bases($feature->get_a_count,$feature->get_c_count,$feature->get_g_count, + $feature->get_t_count,$feature->get_n_count); + $copy->set_conseq($feature->get_match_count,$feature->get_mismatch_count, + $feature->get_unaligned_count); + return($copy); +} + + +# GTF::Feature::offset(ammount) +# This function will offset all locations in this Feature by the given ammount +sub offset{ + my ($feature,$offset) = @_; + unless(defined($offset)){ + print STDERR "Undefined value passed to GTF::Feature::offset.\n"; + return; + } + $feature->{Start} += $offset; + $feature->{End} += $offset; +} + +# GTF::Feature::reverse_complement(seq_length) +# This function takes the length of the sequence this feature came from and +# reverse complements it. +sub reverse_complement{ + my ($feature, $seq_length) = @_; + unless(defined($seq_length)){ + print STDERR + "Undefined value for seq_length passed to ", + "GTF::Feature::reverse_complement.\n"; + } + my $start = $seq_length - $feature->{Start} + 1; + my $stop = $seq_length - $feature->{End} + 1; + if($start < $stop){ + $feature->{Start} = $start; + $feature->{End} = $stop; + } + else{ + $feature->{Start} = $stop; + $feature->{End} = $start; + } +} + +# GTF::Feature::type() +# This function returns the feature type, field of this line as +# a string. Should be either 'CDS', 'exon', 'start_codon', or 'stop_codon'. +sub type {shift->{Type}} + +sub set_type{ + my ($feature,$type) = @_; + $feature->{Type} = $type; +} + +# GTF::Feature::type() +# This function returns the feature sub type, currently inly implemented for +# cds, should be 'Initial', 'Internal', 'Terminal', or 'Single' +sub subtype {shift->{Subtype}} + +sub set_subtype{ + my ($feature,$subtype) = @_; + $feature->{Subtype} = $subtype; +} + +# GTF::Feature::ase() +# This function returns the feature's alternative splicing event, +# if there is one. Currently the only implemented ASE is 'Optional' +# rpz +sub ase {shift->{ASE}} + +sub set_ase { + my ($feature, $ASE) = @_; + $feature->{ASE} = $ASE; +} + +# GTF::Feature::length() +# This function returns the length of this feature in base pairs +# as a number. +sub length{ + my ($feature) = @_; + return($feature->stop - $feature->start + 1); +} + +# GTF::Feature::start() +# This function returns the start location , field, of this feature +# as a number. +sub start {shift->{Start}} + +# GTF::Feature::set_start() +# This function sets the start location , field, of this feature +# as a number. +sub set_start{ + my ($feature,$start) = @_; + $feature->{Start} = $start; +} + +# GTF::Feature::end() +# This function returns the end location, field, of this feature +# as a number. +sub end {shift->{End}} + +# GTF::Feature::stop() +# This function returns the end location, field, of this feature +# as a number. +# Same as GTF::Feature::end(); +sub stop {shift->{End}} + +# GTF::Feature::set_stop() +# This function sets the stop location , field, of this feature +# as a number. +sub set_stop{ + my ($feature,$stop) = @_; + $feature->{End} = $stop; +} + +# GTF::Feature::score() +# This function returns the score, field, of this line. This will +# be either a number or '.'. +sub score {shift->{Score}} + + +# GTF::Feature::frame() +# This function returns the frame, field, of this line. Should be +# '0', '1', '2', or '.'. +sub frame {shift->{Frame}} + +# GTF::Feature::set_frame() +# This function sets the frame, field, of this line. Should be +# '0', '1', '2', or '.'. +sub set_frame{ + my ($feature,$frame) = @_; + if(($frame eq '0') || ($frame eq '1') || + ($frame eq '2') || ($frame eq '.')){ + $feature->{Frame} = $frame; + } + else{ + print STDERR "Bad Frame Value \'$frame\' should be \'0\', \'1\',". + "\'2\', or \'.\'.\n"; + } +} + +# GTF::Feature::transcript_id() +# This function returns the transcript_id of this exon's transcript +sub transcript_id{ + my ($feature) = @_; + my $tx = $feature->{Transcript}; + return $tx->id; +} + +# GTF::Feature::transcript() +# This function returns the transcript of this exon's transcript +sub transcript{ + my ($feature) = @_; + return $feature->{Transcript}; +} + +# GTF::Feature::transcript() +# This function returns the transcript of this exon's transcript +sub set_transcript{ + my ($feature,$tx) = @_; + $feature->{Transcript} = $tx; +} + +# GTF::Feature::transcript_id() +# This function returns the transcript_id of this exon's transcript +sub gene_id{ + my ($feature) = @_; + my $tx = $feature->transcript; + my $gene = $tx->gene; + return $gene->id; +} + +# GTF::Feature::transcript() +# This function returns the transcript of this exon's transcript +sub gene{ + my ($feature) = @_; + my $tx = $feature->{Transcript}; + return $tx->gene; +} + + +# GTF::Feature::seqname() +# This function returns the sequnce name, field, that this +# feature came from as a string. +sub seqname{ + my ($feature) = @_; + my $tx = $feature->{Transcript}; + return $tx->seqname; +} + +# GTF::Feature::source() +# This function returns the source, field, of this line as a +# string. +sub source{ + my ($feature) = @_; + my $tx = $feature->{Transcript}; + return $tx->source; +} + +# GTF::Feature::strand() +# This function returns the strand, field, of this line as a +# string. Should be '+', '-', or '.'. +sub strand{ + my ($feature) = @_; + my $tx = $feature->{Transcript}; + return $tx->strand; +} + +sub equals{ + my ($feature,$compare) = @_; + unless($feature && $compare){ + return 0; + } + if(($feature->start == $compare->start) && + ($feature->stop == $compare->stop) && + ($feature->strand eq $compare->strand) && + ($feature->type eq $compare->type)){ + return 1; + } + return 0; +} + +# GTF::Feature::output_gtf([file_handle]) +# This function outputs the information for this feature in standard +# gtf2 format. If takes and optional file_handle as the only argument +# to which it outputs the data. If no argument is given it outputs +# the data to stdout. +sub output_gtf{ + my ($feature,$out_handle) = @_; + unless(defined $out_handle){ + $out_handle = \*STDOUT; + } + print $out_handle + "".$feature->seqname."\t".$feature->source."\t$feature->{Type}\t", + "$feature->{Start}\t$feature->{End}\t$feature->{Score}\t".$feature->strand."\t", + "$feature->{Frame}\tgene_id \"".$feature->gene_id."\"; transcript_id \"", + $feature->transcript_id."\";\n"; +} + +# GTF::Feature::output_gff([file_handle]) +# This function outputs the information for this feature in standard +# GFF format. If takes and optional file_handle as the only argument +# to which it outputs the data. If no argument is given it outputs +# the data to stdout. +sub output_gff{ + my ($feature,$out_handle) = @_; + unless(defined $out_handle){ + $out_handle = \*STDOUT; + } + + print $out_handle + "".$feature->seqname."\t".$feature->source."\t$feature->{Type}\t", + "$feature->{Start}\t$feature->{End}\t$feature->{Score}\t".$feature->strand."\t", + "$feature->{Frame}\t".$feature->transcript_id."\n"; +} + +# GTF::Feature::set_bases(A,C,G,T,N) +# This funtion allows you to set the number of A,C,G,T nucleotides +# in this feature +sub set_bases{ + my ($feature,$a,$c,$g,$t,$n) = @_; + $feature->{A} = $a; + $feature->{C} = $c; + $feature->{G} = $g; + $feature->{T} = $t; + $feature->{N} = $n; + $feature->{Seq} = 1; +} + +# GTF::Feature::set_conseq(n0,n1,n2) +# This funtion allows you to set the number of 0,1,2 in the +# conservation sequnce for this feature +sub set_conseq{ + my ($feature,$n0,$n1,$n2) = @_; + $feature->{0} = $n0; + $feature->{1} = $n1; + $feature->{2} = $n2; + $feature->{Conseq} = 1; +} + +# GTF::Feature::gc_percentage() +# This funciton returns the gc_percentage for this feature if the +# sequence was loaded otherwise it returns -1. +sub gc_percentage{ + my ($feature) = @_; + if($feature->{GC} != -1){ + return $feature->{GC}; + } + unless($feature->{Seq}){ + return -1; + } + my $a = $feature->{A}; + my $c = $feature->{C}; + my $g = $feature->{G}; + my $t = $feature->{T}; + my $n = $feature->{N}; + my $total = $a + $c + $g + $t + $n; + my $gc = $g + $c; + my $percent = $gc/$total; + $feature->{GC} = $percent; + return $percent; +} + +# GTF::Feature::match_percentage() +# This funciton returns the percentage of matches in the conservation sequence +# for this feature if it was loaded otherwise it returns -1. +sub match_percentage{ + my ($feature) = @_; + if($feature->{Match} != -1){ + return $feature->{Match}; + } + unless($feature->{Conseq}){ + return -1; + } + my $match = $feature->{1}; + my $mismatch = $feature->{0}; + my $unaligned = $feature->{2}; + my $total = $match + $mismatch + $unaligned; + my $percent = $match/$total; + $feature->{Match} = $percent; + return $percent; +} + +# GTF::Feature::mismatch_percentage() +# This funciton returns the percentage of mismatches in the conservation sequence +# for this feature if it was loaded otherwise it returns -1. +sub mismatch_percentage{ + my ($feature) = @_; + if($feature->{Mismatch} != -1){ + return $feature->{Mismatch}; + } + unless($feature->{Conseq}){ + return -1; + } + my $match = $feature->{1}; + my $mismatch = $feature->{0}; + my $unaligned = $feature->{2}; + my $total = $match + $mismatch + $unaligned; + my $percent = $mismatch/$total; + $feature->{Mismatch} = $percent; + return $percent; +} + +# GTF::Feature::unaligned_percentage() +# This funciton returns the percentage of unaligned nucs in the conservation sequence +# for this feature if it was loaded otherwise it returns -1. +sub unaligned_percentage{ + my ($feature) = @_; + if($feature->{Unaligned} != -1){ + return $feature->{Unaligned}; + } + unless($feature->{Conseq}){ + return -1; + } + my $match = $feature->{1}; + my $mismatch = $feature->{0}; + my $unaligned = $feature->{2}; + my $total = $match + $mismatch + $unaligned; + my $percent = $unaligned/$total; + $feature->{Unaligned} = $percent; + return $percent; +} + +# GTF::Feature::get_a_count() +# This funciton returns the number of A nucleotides in this features sequence +# if the sequence was loaded, otherwise it returns -1. +sub get_a_count{ + my ($feature) = @_; + unless($feature->{Seq}){ + return -1; + } + return $feature->{A}; +} + +# GTF::Feature::get_c_count() +# This funciton returns the number of C nucleotides in this features sequence +# if the sequence was loaded, otherwise it returns -1. +sub get_c_count{ + my ($feature) = @_; + unless($feature->{Seq}){ + return -1; + } + return $feature->{C}; +} + +# GTF::Feature::get_g_count() +# This funciton returns the number of G nucleotides in this features sequence +# if the sequence was loaded, otherwise it returns -1. +sub get_g_count{ + my ($feature) = @_; + unless($feature->{Seq}){ + return -1; + } + return $feature->{G}; +} + +# GTF::Feature::get_t_count() +# This funciton returns the number of T nucleotides in this features sequence +# if the sequence was loaded, otherwise it returns -1. +sub get_t_count{ + my ($feature) = @_; + unless($feature->{Seq}){ + return -1; + } + return $feature->{T}; +} + +# GTF::Feature::get_n_count() +# This funciton returns the number of N nucleotides in this features sequence +# if the sequence was loaded, otherwise it returns -1. +sub get_n_count{ + my ($feature) = @_; + unless($feature->{Seq}){ + return -1; + } + return $feature->{N}; +} + +# GTF::Feature::get_match_count() +# This funciton returns the number of matched nucleotides in this features +# conservation sequence if it was loaded, otherwise it returns -1. +sub get_match_count{ + my ($feature) = @_; + unless($feature->{Conseq}){ + return -1; + } + return $feature->{1}; +} + +# GTF::Feature::get_mismatch_count() +# This funciton returns the number of mismatched nucleotides in this features +# conservation sequence if it was loaded, otherwise it returns -1. +sub get_mismatch_count{ + my ($feature) = @_; + unless($feature->{Conseq}){ + return -1; + } + return $feature->{0}; +} + +# GTF::Feature::get_unaligned_count() +# This funciton returns the number of unaligned nucleotides in this features +# conservation sequence if it was loaded, otherwise it returns -1. +sub get_unaligned_count{ + my ($feature) = @_; + unless($feature->{Conseq}){ + return -1; + } + return $feature->{2}; +} + +sub tag{ + my ($feature) = @_; + return $feature->{Tag}; +} + +sub set_tag{ + my ($feature,$tag) = @_; + $feature->{Tag} = $tag; +} + +1; +__END__ diff --git a/99.scripts/trinity_utils/PerlLib/GTF_utils.pm b/99.scripts/trinity_utils/PerlLib/GTF_utils.pm new file mode 100644 index 0000000..ef0e601 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/GTF_utils.pm @@ -0,0 +1,392 @@ +package GTF_utils; + +use strict; +use warnings; +use Gene_obj; +use Gene_obj_indexer; +use GTF; +use Carp; +use Data::Dumper; +use Overlap_piler; + +#### +sub index_GTF_gene_objs { + my ($gtf_filename, $gene_obj_indexer) = @_; ## gene_obj_indexer can be a simple hashref {} + + unless ($gtf_filename && $gene_obj_indexer) { + confess "Error, need gtf_filename and gene_obj_indexer as perams"; + } + + return index_GTF_gene_objs_from_GTF($gtf_filename,$gene_obj_indexer); +} + +sub index_GTF_gene_objs_from_GTF { + my ($gtf_filename, $gene_obj_indexer) = @_; + + ## + #print STDERR "\n-caching genes.\n"; + my %seqname_map; + + my $gene_objs = GTF_to_gene_objs($gtf_filename); + + my %seen; + + for my $gene_obj (@$gene_objs) { + + my $gene_id = $gene_obj->{TU_feat_name}; + + + if ($seen{$gene_id}) { + confess "Error, already processed gene: $gene_id\n" + . " here: " . $gene_obj->toString() . "\n" + . " and earlier: " . $seen{$gene_id}->toString(); + + } + + $seen{$gene_id} = $gene_obj; + + + my $seqname = $gene_obj->{asmbl_id}; + + if (ref $gene_obj_indexer eq "HASH") { + $gene_obj_indexer->{$gene_id} = $gene_obj; + } + else { + $gene_obj_indexer->store_gene($gene_id, $gene_obj) if(ref $gene_obj_indexer); + } + + # add to gene list for asmbl_id + my $gene_list = $seqname_map{$seqname}; + unless (ref $gene_list) { + $gene_list = $seqname_map{$seqname} = []; + } + push (@$gene_list, $gene_id); + } + return (\%seqname_map); +} + +sub GTF_to_gene_objs { + my ($gtf_filename) = @_; + + my %gene_transcript_data; + my %noncoding_features; + + my %gene_id_to_source; + my %gene_id_to_name; + + my %gene_id_to_seq_name; + my %gene_id_to_gene_name; + + my %gene_id_to_gene_type; + my %transcript_id_to_transcript_type; + + + my %coding_genes; + + + open (my $fh, $gtf_filename) or die "Error, cannot open $gtf_filename"; + while (<$fh>) { + unless (/\w/) { next; } + if (/^\#/) { next; } # comment line. + + chomp; + my ($seqname, $source, $type, $lend, $rend, $score, $strand, $gtf_phase, $annot) = split (/\t/); + + my ($end5, $end3) = ($strand eq '+') ? ($lend, $rend) : ($rend, $lend); + + $annot =~ /gene_id \"([^\"]+)\"/ or confess "Error, cannot get gene_id from $annot of line\n$_"; + my $gene_id = $1; + + if (my $sn = $gene_id_to_seq_name{$gene_id}) { + if ($sn ne $seqname) { + $gene_id = "$seqname" . "|" . "$gene_id"; # make unique per scaffold. + } + } + $gene_id_to_seq_name{$gene_id} = $seqname; + + $gene_id_to_source{$gene_id} = $source; + + + if ($annot =~ /name \"([^\"]+)\"/) { + my $name = $1; + $gene_id_to_name{$gene_id} = $name; + } + + my $gene_name = ""; + if ($annot =~ /gene_name \"([^\"]+)\"/) { + $gene_name = $1; + $gene_id_to_gene_name{$gene_id} = $gene_name; + } + + if ($annot =~ /gene_type \"([^\"]+)\"/) { + my $gene_type = $1; + $gene_id_to_gene_type{$gene_id} = $gene_type; + } + + + # print "gene_id: $gene_id, transcrpt_id: $transcript_id, $type\n"; + + if ($type eq 'transcript' || $type eq 'gene') { next; } # capture by exon coordinates + + my $transcript_id; + if ($annot =~ /transcript_id \"([^\"]+)\"/) { + $transcript_id = $1; + + if ($annot =~ /transcript_type \"([^\"]+)\"/) { + my $transcript_type = $1; + $transcript_id_to_transcript_type{$transcript_id} = $transcript_type; + } + } + else { + print STDERR "Skipping line: $_, no transcript_id value provided\n"; + next; + } + + if ($type eq 'CDS' || $type eq 'stop_codon' || $type eq 'start_codon') { + push (@{$gene_transcript_data{$seqname}->{$gene_id}->{$transcript_id}->{CDS}}, [$end5, $end3] ); + push (@{$gene_transcript_data{$seqname}->{$gene_id}->{$transcript_id}->{mRNA}}, [$end5, $end3] ); + + $coding_genes{$gene_id}++; + } + elsif ($type eq "exon" || $type =~ /UTR/) { + push (@{$gene_transcript_data{$seqname}->{$gene_id}->{$transcript_id}->{mRNA}}, [$end5, $end3] ); + } + elsif ($type =~ /Selenocysteine/) { + # no op + } + else { + ## assuming noncoding feature + push (@{$noncoding_features{$seqname}->{$type}->{$gene_id}->{$transcript_id}}, [$end5, $end3] ); + } + + } + close $fh; + + + ## create gene objects. + + my @top_gene_objs; + + my %seen; + foreach my $seqname (keys %gene_transcript_data) { + + + { + ################################## + ## Process protein-coding genes: + + my $genes_href = $gene_transcript_data{$seqname}; + + foreach my $gene_id (keys %$genes_href) { + + if ($seen{$gene_id}) { + print STDERR ("Error, already saw $gene_id,$seqname as $gene_id,$seen{$gene_id}\nSkipping."); + next; + } + $seen{$gene_id} = $seqname; + + my $transcripts_href = $genes_href->{$gene_id}; + + my $source = $gene_id_to_source{$gene_id}; + + my @gene_objs; + + foreach my $transcript_id (keys %$transcripts_href) { + + my $coord_types_href = $transcripts_href->{$transcript_id}; + + my $CDS_coords_aref = $coord_types_href->{CDS}; + my $mRNA_coords_aref = $coord_types_href->{mRNA}; + + + #print STDERR "Before, CDS: " . Dumper($CDS_coords_aref); + #print STDERR "Before, exons: " . Dumper($mRNA_coords_aref); + + + my $CDS_coords_href = &_join_overlapping_coords($CDS_coords_aref); + my $mRNA_coords_href = &_join_overlapping_coords($mRNA_coords_aref); + + #print STDERR "CDS: " . Dumper($CDS_coords_href); + #print STDERR "mRNA: " . Dumper ($mRNA_coords_href); + + + my $gene_obj = new Gene_obj(); + $gene_obj->populate_gene_object($CDS_coords_href, $mRNA_coords_href); + + $gene_obj->{TU_feat_name} = $gene_id; + $gene_obj->{Model_feat_name} = $transcript_id; + if (my $name = $gene_id_to_name{$gene_id}) { + $gene_obj->{com_name} = $name; + } + else { + $gene_obj->{com_name} = $transcript_id; + } + if (my $gene_name = $gene_id_to_gene_name{$gene_id}) { + $gene_obj->{gene_name} = $gene_name; + } + $gene_obj->{asmbl_id} = $seqname; + $gene_obj->{source} = $source; + + if (my $gene_type = $gene_id_to_gene_type{$gene_id}) { + $gene_obj->{gene_type} = $gene_type; + } + if (my $transcript_type = $transcript_id_to_transcript_type{$transcript_id}) { + $gene_obj->{transcript_type} = $transcript_type; + } + + $gene_obj->join_adjacent_exons(); + + push (@gene_objs, $gene_obj); + } + + + ## want single gene that includes all alt splice variants here + if(scalar(@gene_objs)) { + my $template_gene_obj = shift @gene_objs; + foreach my $other_gene_obj (@gene_objs) { + $template_gene_obj->add_isoform($other_gene_obj); + } + push (@top_gene_objs, $template_gene_obj); + + # print $template_gene_obj->toString(); + + + } + } + + } + + + { + ################################ + ## Process noncoding features ## + ################################ + + my $ncgene_types_href = $noncoding_features{$seqname}; + + if (ref $ncgene_types_href) { + + foreach my $nc_type (keys %$ncgene_types_href) { + + my $gene_ids_href = $ncgene_types_href->{$nc_type}; + foreach my $gene_id (keys %$gene_ids_href) { + + if (exists $coding_genes{$gene_id}) { + print STDERR "Warning: Skipping $gene_id ($nc_type) as this gene is already included as a coding gene.\n"; + next; + } + + my $trans_ids_href = $gene_ids_href->{$gene_id}; + foreach my $trans_id (keys %$trans_ids_href) { + + my @coordsets = @{$trans_ids_href->{$trans_id}}; + my %coords; + foreach my $coordset (@coordsets) { + my ($end5, $end3) = @$coordset; + $coords{$end5} = $end3; + } + + my $gene_obj = new Gene_obj(); + + $gene_obj->populate_gene_object({}, \%coords); + $gene_obj->{asmbl_id} = $seqname; + $gene_obj->{TU_feat_name} = $gene_id; + $gene_obj->{Model_feat_name} = $trans_id; + $gene_obj->{gene_type} = $nc_type; + if (my $name = $gene_id_to_name{$gene_id}) { + $gene_obj->{com_name} = $name; + } + else { + $gene_obj->{com_name} = $trans_id; + } + + + #print STDERR $gene_obj->toString(); + + + push (@top_gene_objs, $gene_obj); + } + } + } + } + } + } + return (\@top_gene_objs); +} + + + +#### +sub _join_overlapping_coords { + my $coords_aref = shift; + + unless (ref $coords_aref) { + return ({}); + } + + my $orient; + + my @coords; + + my $inferred_orient; + + foreach my $coordset (@$coords_aref) { + my ($end5, $end3) = @$coordset; + + my $orient; + if ($end5 < $end3) { + $orient = '+'; + } + elsif ($end5 > $end3) { + $orient = '-'; + + } + + if ( (! defined $inferred_orient) && defined $orient) { + $inferred_orient = $orient; + } + elsif ( defined($orient) && $orient ne $inferred_orient) { + die "Error, conflicting orientation info: " . Dumper($coords_aref); + } + + my ($lend, $rend) = sort {$a<=>$b} ($end5, $end3); + + push (@coords, [$lend, $rend]); + } + + #print STDERR "coords: " . Dumper(@coords); + + my @piles = &Overlap_piler::simple_coordsets_collapser(@coords); + + #print STDERR "piles: " . Dumper(@piles); + + @piles = sort {$a->[0] <=> $b->[0]} @piles; + + ## join adjacent piles + my @joined_piles = shift @piles; + foreach my $pile (@piles) { + my ($pile_lend, $pile_rend) = @$pile; + if ($pile_lend == $joined_piles[$#joined_piles]->[1] + 1) { + # adjacent + $joined_piles[$#joined_piles]->[1] = $pile_rend; + } + else { + push (@joined_piles, $pile); + } + } + + + my %new_coords; + foreach my $pile (@joined_piles) { + my ($lend, $rend) = @$pile; + my ($end5, $end3) = ($inferred_orient eq '+') ? ($lend, $rend) : ($rend, $lend); + $new_coords{$end5} = $end3; + } + + + return(\%new_coords); +} + + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/Gene_obj.pm b/99.scripts/trinity_utils/PerlLib/Gene_obj.pm new file mode 100644 index 0000000..12fc005 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Gene_obj.pm @@ -0,0 +1,5587 @@ +#!/usr/bin/env perl + +package main; +our $DEBUG; + +package Gene_obj; +use strict; +use Nuc_translator; +#use Gene_ontology; +use Longest_orf; +use Storable qw (store retrieve freeze thaw dclone); +use warnings; +use Data::Dumper; +use Carp qw (croak cluck confess); +#use URI::Escape; + +=head1 NAME + +package Gene_obj + +=cut + + + +=head1 DESCRIPTION + + Gene_obj(s) encapsulate the elements of both gene structure and gene function. The gene structure is stored in a hierarchical fashion as follows: + + Gene ========================================================= + + Exon ========= ========= ========= ======== + + CDS ====== ========= ====== + + + where a Gene is a container for Exon(s), and each Exon is a container for a CDS, and an Exon can contain a single CDS component. An Exon lacking a CDS exon is an untranslated exon or UTR exon. The region of an Exon which extends beyond the CDS is also considered a UTR. + + + There are several ways to instantiate gene objects. A simple example is described: + + Exon and CDS component coordinates can be assigned as hashes. + + ie. + + my %mrna = ( 100 => 200, + 300 => 500 ); + + my %CDS = ( 150=>200, + 300=>450); + + my $sequence = "GACTACATTTAATAGGGCCC"; #string representing the genomic sequence + my $gene = new Gene_obj(); + + $gene->{com_name} = "hypothetical protein"; + + $gene->populate_gene_obj(\%CDS, \%mRNA, \$sequence); + print $gene->toString(); + + + + Alternatively, the individual components of genes (Exons and CDSs) can be instantiated separately and used to build the Gene from the ground up (See packages mRNA_exon_obj and CDS_exon_obj following this Gene_obj documentation). + + my $cds_exon = new CDS_exon_obj (150, 200); + + my $mRNA_exon = new mRNA_exon_obj (100, 200); + + $mRNA_exon->set_CDS_exon_obj($cds_exon); + + my $gene_obj = new Gene_obj (); + + $gene_obj->{gene_name} = "hypothetical gene"; + $gene_obj->{com_name} = "hypothetical protein"; + + $gene_obj->add_mRNA_exon_obj($mRNA_exon); + + $gene_obj->refine_gene_object(); + + $gene_obj->create_all_sequence_types (\$sequence); #ref to genomic sequence string. + + print $gene_obj->toString(); + + + The API below describes useful functions for navigating and manipulating the Gene object along with all of its attributes. + + + +=cut + + + + + + +=over 4 + +=item new() + +B Constructor for Gene_obj + +B none + +B $gene_obj + + +The Gene_obj contains several attributes which can be manipulated directly (or by get/set methods if they exist). These attributes include: + + asmbl_id # identifier for the genomic contig for which this gene is anchored. + TU_feat_name #feat_names are TIGR temporary identifiers. + Model_feat_name # temp TIGR identifier for gene models + locus #identifier for a gene (TU) ie. T2P3.5 + pub_locus #another identifier for a gene (TU) ie. At2g00010 + model_pub_locus #identifier for a gene model (model) ie. At2g00010.1 + model_locus #analagous to locus, but for model rather than gene (TU) + alt_locus #alternative locus + gene_name # name for gene + com_name # name for gene product + comment #internal comment + pub_comment #comment related to gene + ec_num # enzyme commission number + gene_sym # gene symbol + is_5prime_partial # 0|1 missing start codon. + is_3prime_partial # 0|1 missing stop codon. + is_pseudogene # 0|1 + curated_com_name # 0|1 + curated_gene_structure # 0|1 + + ## Other attributes set internally Access-only, do not set directly. + + gene_length # length of gene span (int). + mid_pt # holds midpoint of gene-span + strand # [+-] + protein_seq # holds protein sequence + protein_seq_length + CDS_sequence #holds CDS sequence (translated to protein); based on CDS_exon coordinates + CDS_seq_length + cDNA_sequence #holds cDNA sequence; based on mRNA exon coordinates. + cDNA_seq_length + gene_sequence #holds unspliced transcript + gene_sequence_length #length of unspliced transcript + gene_type # "protein-coding", #default type for gene object. Could be changed to "rRNA|snoRNA|snRNA|tRNA" to accommodate other gene or feature types. + num_additional_isoforms # int + + +=back + +=cut + + + +sub new { + shift; + my $self = { asmbl_id => 0, #genomic contig ID + locus => undef, #text + pub_locus => undef, #text ie. At2g00010 + model_pub_locus =>undef, #text ie. At2g00010.1 + model_locus => undef, #text ie. F12G15.1 + alt_locus => undef, #text + gene_name => undef, #text + com_name => undef, #text + comment => undef, + curated_com_name => 0, + curated_gene_structure => 0, + pub_comment => undef, #text + ec_num => undef, #text (enzyme commission number) + gene_type => "protein-coding", #default type for gene object. Could be changed to "rRNA|snoRNA|snRNA|tRNA" to accomodate other gene or feature types. + gene_sym => undef, #text (gene symbol) + mRNA_coords => 0, #assigned to anonymous hash of end5->end3 relative to the parent sequence + CDS_coords => 0, #assigned to anonymous hash of end5->end3 relative to the parent sequence + mRNA_exon_objs => 0, # holds arrayref to mRNA_obj, retrieve only thru method: get_exons() + num_exons => 0, # number of exons in this gene_obj + model_span => [], # holds array ref to (end5,end3) for CDS range of gene. + gene_span => [], # holds array ref to (end5,end3) for mRNA range of gene. + gene_length => 0, # length of gene span (int). + mid_pt => 0, # holds midpoint of gene-span + strand => 0, # [+-] + gi => undef, #text + prot_acc => undef, #text + is_pseudogene => 0, # toggle indicating pseudogene if 1. + is_5prime_partial => 0, #boolean indicating missing 5' part of gene. + is_3prime_partial => 0, #boolean + protein_seq => undef, # holds protein sequence + protein_seq_length => 0, + CDS_sequence => undef, #holds CDS sequence (translated to protein); based on CDS_exon coordinates + CDS_seq_length => 0, + cDNA_sequence => undef, #holds cDNA sequence; based on mRNA exon coordinates. + cDNA_seq_length => 0, + gene_sequence => undef, #holds unspliced transcript + gene_sequence_length => 0, #length of unspliced transcript + TU_feat_name => undef, #feat_names are TIGR temporary identifiers. + Model_feat_name =>undef, + classification => 'annotated_genes', #type of seq_element. + gene_synonyms => [], #list of synonymous model feat_names + GeneOntology=>[], #list of Gene_ontology assignment objects. ...see GeneOntology.pm + + ## Additional functional attributes: + secondary_gene_names => [], + secondary_product_names => [], + secondary_gene_symbols => [], + secondary_ec_numbers =>[], + + + ## Alternative splicing support. + num_additional_isoforms => 0, # number of additional isoforms stored in additonal_isoform list below + additional_isoforms => [] # stores list of Gene_objs corresponding to the additional isoforms. + + }; + bless($self); + return ($self); +} + + + + +=over 4 + +=item erase_gene_structure() + +B Removes the structural components of a gene (ie. exons, CDSs, coordinate spans, any corresponding sequences) + +B none + +B none + +=back + +=cut + + +## erase gene structure +sub erase_gene_structure { + my $self = shift; + $self->{mRNA_exon_objs} = 0; + $self->{num_exons} = 0; + $self->{model_span} = []; + $self->{gene_span} = []; + $self->{gene_length} = 0; + $self->{strand} = 0; + $self->{protein_seq} = 0; + $self->{CDS_sequence} = 0; + $self->{CDS_seq_length} = 0; + $self->{cDNA_sequence} = 0; + $self->{cDNA_seq_length} = 0; +} + + +=over 4 + +=item clone_gene() + +B Clones this Gene_obj by copying attributes from this Gene to a new gene. Does NOT do a deep clone for all attributes. See dclone() for a more rigorous cloning method. This method is safer because all references are not cloned, only the critical ones. + +B none + +B new Gene_obj + +=back + +=cut + + + +## all objects are cloned. References to data only are not. +sub clone_gene { + my $self = shift; + my $clone = new Gene_obj(); + + + ## Copy over the non-ref attribute values. + foreach my $key (keys %$self) { + my $value = $self->{$key}; + if (defined $value) { + ## Not copying over refs. + if (ref $value) { + next; + } + + ## Not copying over attributes of length > 200, such as protein/nucleotide sequences + my $length = length($value); + if ($length > 200) { next;} + } + + # passed tests above, copying attribute. + $clone->{$key} = $value; + + } + + ## copy over the gene synonyms. + my @gene_syns = @{$self->{gene_synonyms}}; + $clone->{gene_synonyms} = \@gene_syns; + + + ## copy the GO assignments: + my @GO_assignments = $self->get_gene_ontology_objs(); + if (@GO_assignments) { + foreach my $go_assignment (@GO_assignments) { + my $go_clone = dclone($go_assignment); + $clone->add_gene_ontology_objs($go_clone); + } + } + + + ## copy gene structure. + my @exons = $self->get_exons(); + foreach my $exon (@exons) { + $clone->add_mRNA_exon_obj($exon->clone_exon()); + } + + foreach my $isoform ($self->get_additional_isoforms()) { + my $isoform_clone = $isoform->clone_gene(); + $clone->add_isoform($isoform_clone); + } + + $clone->refine_gene_object(); + + return ($clone); +} + + + + +=over 4 + +=item deep_clone() + +B Provides a deep clone of a gene_obj. Only references supported in Gene_obj documentation are supported. Those added in a rogue way are undef()d + +B none + +B $gene_obj + +uses the Storable dclone() function to deep clone the Gene_obj + +=back + +=cut + + + ; +## all objects are cloned. References to data only are not. +sub deep_clone { + my $self = shift; + my $clone = dclone($self); + + my %supported_refs = (model_span => 1, + gene_span => 1, + gene_synonyms => 1, + Gene_ontology => 1, + additional_isoforms=>1, + mRNA_exon_objs => 1); + + foreach my $gene_obj ($clone, $clone->get_additional_isoforms()) { + + my @keys = keys %$gene_obj; + foreach my $key (@keys) { + my $value = $gene_obj->{$key}; + if (ref $value && !$supported_refs{$key}) { + $gene_obj->{$key} = undef; + } + } + } + + return ($clone); +} + + +=over 4 + +=item populate_gene_obj() + +B Given CDS and mRNA coordinates stored in hash form, a gene object is populated with mRNA and CDS exons. This is one available way to populate a newly instantiated Gene_obj. + +B $cds_hash_ref, $mRNA_hash_ref, <$seq_ref> + +$mRNA_hash_ref is a reference to a hash holding the end5 => end3 coordinates of the Exons + +$cds_hash_ref same as mRNA_has_ref except holds the CDS end5 => end3 coordinates. + +$seq_ref is a reference to a string containing the genomic sequence. This is an optional parameter. + + +B none + +=back + +=cut + + ; + +## Do several things at once: assign CDS and mRNA coordinates, and build gene sequences. +## The \$seq_ref is optional in case you want to create the sequence types. +sub populate_gene_obj { + my ($self, $cds_ref, $mRNA_ref, $seq_ref) = @_; + $self->set_CDS_coords ($cds_ref); + $self->set_mRNA_coords ($mRNA_ref); + $self->refine_gene_object(); + if (ref $seq_ref) { + $self->create_all_sequence_types($seq_ref); + } + ## reinitialize the hashrefs: + $self->{mRNA_coords} = 0; + $self->{CDS_coords} = 0; + + +} + + +# alias above +sub populate_gene_object { + my $self = shift; + $self->populate_gene_obj(@_); +} + + +#### +sub populate_gene_object_via_CDS_coords { + my $self = shift; + my @coordsets = @_; + + foreach my $coordset (@coordsets) { + my ($end5, $end3) = @$coordset; + my $mrna_exon_obj = mRNA_exon_obj->new($end5, $end3); + my $cds_obj = CDS_exon_obj->new($end5, $end3); + $mrna_exon_obj->{CDS_exon_obj} = $cds_obj; + $self->add_mRNA_exon_obj($mrna_exon_obj); + } + + $self->refine_gene_object(); + return; +} + + +sub build_gene_obj_exons_n_cds_range { + my $self = shift; + my ($exons_aref, $cds_lend, $cds_rend, $orient) = @_; + + my @exon_coords; + foreach my $exon_aref (@$exons_aref) { + my ($exon_lend, $exon_rend) = sort {$a<=>$b} @$exon_aref; + push (@exon_coords, [$exon_lend, $exon_rend] ); + } + @exon_coords = sort {$a->[0]<=>$b->[0]} @exon_coords; + + + unless ($orient =~ /^[\+\-]$/) { + confess "Error, orient not [+-] "; + } + + ## build the CDS coordinates. + + my @cds_range; + + if ($cds_lend > 0 && $cds_rend > 0) { + + ($cds_lend, $cds_rend) = sort {$a<=>$b} ($cds_lend, $cds_rend); + + foreach my $exon_coords_aref (@exon_coords) { + my ($exon_lend, $exon_rend) = @$exon_coords_aref; + + if ($exon_rend >= $cds_lend && $exon_lend <= $cds_rend) { + + ## got overlap + my $cds_exon_lend = ($cds_lend < $exon_lend) ? $exon_lend : $cds_lend; + + my $cds_exon_rend = ($cds_rend > $exon_rend) ? $exon_rend : $cds_rend; + + push (@cds_range, [$cds_exon_lend, $cds_exon_rend]); + } + } + + unless (@cds_range) { + confess "Error, no CDS exon coords built based on exon overlap"; + } + } + ## all coordinate sets are ordered left to right. + # build the coordinates href + + my %exon_coords; + my %cds_coords; + foreach my $exon_coords_aref (@exon_coords) { + my ($exon_lend, $exon_rend) = @$exon_coords_aref; + my ($exon_end5, $exon_end3) = ($orient eq '+') ? ($exon_lend, $exon_rend) : ($exon_rend, $exon_lend); + $exon_coords{$exon_end5} = $exon_end3; + } + foreach my $cds_coords_aref (@cds_range) { + my ($cds_lend, $cds_rend) = @$cds_coords_aref; + my ($cds_end5, $cds_end3) = ($orient eq '+') ? ($cds_lend, $cds_rend) : ($cds_rend, $cds_lend); + $cds_coords{$cds_end5} = $cds_end3; + } + + # print Dumper (\%cds_coords) . Dumper (\%exon_coords); + + $self->populate_gene_obj(\%cds_coords, \%exon_coords); + + return ($self); +} + + +#### +sub join_adjacent_exons { + my $self = shift; + + my @exons = $self->get_exons(); + + my $strand = $self->get_orientation(); + + my $first_exon = shift @exons; + my @new_exons = ($first_exon); + + while (@exons) { + my $prev_exon = $new_exons[$#new_exons]; + my ($prev_end5, $prev_end3) = $prev_exon->get_coords(); + + my $next_exon = shift @exons; + my ($next_end5, $next_end3) = $next_exon->get_coords(); + + if ( ($strand eq '+' && $prev_end3 == $next_end5 - 1) # adjacent + || + ($strand eq '-' && $prev_end3 == $next_end5 + 1) ) { + + $prev_exon->merge_exon($next_exon); + } + else { + push (@new_exons, $next_exon); + } + } + + $self->{mRNA_exon_objs} = [@new_exons]; + + $self->refine_gene_object(); + + return; +} + + + + +=over 4 + +=item AAToNucleotideCoords() + +B Converts an amino acid -based coordinate to a genomic sequence -based coordinate. + +B $aa_coord + +B $genomic_coord + +undef is returned if the aa_coord could not be converted. + + +=back + +=cut + + ; + +sub AAToNucleotideCoords{ + my($self) = shift; + my($aacoord) = shift; + my($debug) = shift; + my($PCDS_coords) = {}; + my($A2NMapping) = {}; + my($currAA) = 1; + my $strand = $self->{strand}; + my @exons = $self->get_exons(); + my($cds_count)=0; + my($translated_bp)=-1; + my($lastcarryover)=0; + my($end_bp); + foreach my $exon (sort { + if($strand eq "+"){ + $a->{end5}<=>$b->{end5}; + } + else{ + $b->{end5}<=>$a->{end5}; + } + } @exons) { + my $cds = $exon->get_CDS_obj(); + if ($cds) { + my @cds_coords = $cds->get_CDS_end5_end3(); + my($bpspread) = abs($cds_coords[0]-$cds_coords[1]); + $bpspread+=$lastcarryover; + my($nextAA) = int($bpspread/3); # last complete AA in CDS + $lastcarryover = $bpspread%3; + $PCDS_coords->{$currAA} = $currAA+$nextAA-1; + if($strand eq "+"){ + $A2NMapping->{$currAA} = $cds_coords[0]<$cds_coords[1]?$cds_coords[0]:$cds_coords[1]; + } + else{ + $A2NMapping->{$currAA} = $cds_coords[0]<$cds_coords[1]?$cds_coords[1]:$cds_coords[0]; + } + print "DEBUG: $strand $cds_count AA range ($currAA - $PCDS_coords->{$currAA}) nucleotide start($A2NMapping->{$currAA})\n" if($debug); + $currAA = $currAA+$nextAA; + $cds_count++; + if($strand eq "+"){ + $end_bp = $cds_coords[0]<$cds_coords[1]?$cds_coords[1]:$cds_coords[0]; + } + else{ + $end_bp = $cds_coords[0]<$cds_coords[1]?$cds_coords[0]:$cds_coords[1]; + } + } + } + # PCDS_coords key/value are start/stop aa counts for each cds; + # A2NMapping stores cds AA start key to cds nucleotide start + $cds_count=0; + foreach my $PCDS_end5 (sort { + $a<=>$b; + }(keys %$PCDS_coords)) { + my($PCDS_end3) = $PCDS_coords->{$PCDS_end5}; + if($aacoord>=$PCDS_end5 && $aacoord<=$PCDS_end3){ + my($nucleotide_start) = $A2NMapping->{$PCDS_end5}; + my($aa_offset) = $aacoord - $PCDS_end5; + my($nucleotide_offset) = $aa_offset*3; + print "DEBUG: CDS offset $aa_offset AA $nucleotide_offset bp\n" if($debug); + if($strand eq "+"){ + $translated_bp = $nucleotide_start+$nucleotide_offset; + } + else{ + $translated_bp = $nucleotide_start-$nucleotide_offset; + } + print "DEBUG: Mapping $aacoord to $translated_bp in cds $cds_count\n" if($debug); + print "DEBUG: CDS $PCDS_end5 - $PCDS_end3 nucleotide start $A2NMapping->{$PCDS_end5}, nuc offset $nucleotide_offset\n" if($debug); + } + + $cds_count++; + } + #} + if($translated_bp == -1){ + $translated_bp = undef; + print STDERR "Unable to translate AA coordinate: $aacoord. Off end. Using undef\n" if($debug); + } + return $translated_bp; +} + + + +## private method, used by populate_gene_obj() +# sets CDS_coords instance member to a hash reference of CDS coordinates. $hash{end5} = end3 +sub set_CDS_coords { + my $self = shift; + my $hash_ref = shift; + if (ref ($hash_ref) eq 'HASH') { + $self->{CDS_coords} = $hash_ref; + } else { + print STDERR "Cannot set CDS_coords, must have hash reference\n"; + } +} + + + + +=over 4 + +=item get_gene_span() + +B Retrieves the coordinates which span the length of the gene along the genomic sequence. + +B none + +B (end5, end3) + +These coordinates represent the minimal and maximal exonic coordinates of the gene. Orientation can be inferred by the relative values of end5 and end3. + + +=back + +=cut + + ; + +## All return gene end5, end3 ### +sub get_gene_span { + my $self = shift; + return (@{$self->{gene_span}}); +} + + + + +## private +sub get_seq_span { + my $self = shift; + return ($self->get_gene_span()); +} + + + +=over 4 + +=item get_coords() + +B See get_gene_span() + +B none + +B (end5, end3) + +=back + +=cut + + +sub get_coords { + my $self = shift; + return ($self->get_gene_span()); +} + + + +=over 4 + +=item get_model_span() + +B Retrieves the coordinates spanned by the protein-coding region of the gene along the genomic sequence. + +B none + +B (end5, end3) + +These coordinates are determined by the min and max of the CDS components of the gene. + +=back + +=cut + + + + +sub get_model_span { + my $self = shift; + return (@{$self->{model_span}}); +} + + +sub get_CDS_span { # preferred + my $self = shift; + return($self->get_model_span()); +} + + +=over 4 + +=item get_transcript_span() + +B Retrieves the coordinates spanned by the exonic regions of the gene along the genomic sequence. + +B none + +B (lend, rend) + +These coordinates are determined by the min and max of the CDS components of the gene. + +=back + +=cut + + +sub get_transcript_span { + my $self = shift; + + my @coords; + my @exons = $self->get_exons(); + foreach my $exon (@exons) { + push (@coords, $exon->get_coords()); + } + @coords = sort {$a<=>$b} @coords; + + my $lend = shift @coords; + my $rend = pop @coords; + + return($lend, $rend); +} + + +sub is_pseudogene { + my $self = shift; + return ($self->{is_pseudogene}); +} + +sub set_pseudogene { + my $self = shift; + my $pseudogene_val = shift; + unless ($pseudogene_val =~ /[01]/) { + confess "Error, can set pseudogene to zero or one only.\n"; + } + + foreach my $gene ($self, $self->get_additional_isoforms()) { + $gene->{is_pseudogene} = $pseudogene_val; + } + + return; +} + + + +#private +# sets mRNA_coords instance member to a hash reference of CDS coordinates. $hash{end5} = end3 +sub set_mRNA_coords { + my $self = shift; + my $hash_ref = shift; + if (ref ($hash_ref) eq 'HASH') { + $self->{mRNA_coords} = $hash_ref; + } else { + print STDERR "Cannot set CDS_coords, must have hash reference\n"; + } +} + + +=over 4 + +=item refine_gene_object() + +B This method performs some data management operations and should be called at any time modifications have been made to the gene structure (ie. exons added or modified, model isoforms added, etc). It performs the following orientations: + + -Sets (or resets) gene span and model span coordinates, strand orientation, gene length, mid-point. + +B none + +B none + +=back + +=cut + +## Once mRNA_coords and CDS_coords have been assigned, this will populate the remaining elements in the gene object. + +sub refine_gene_object { + my ($self) = shift; + #check to see if mRNA_coords field is populated. If not, initialize. + if ($self->{mRNA_coords} == 0) { + $self->{mRNA_coords} = {}; + } + my ($CDS_coords, $mRNA_coords) = ($self->{CDS_coords}, $self->{mRNA_coords}); + + unless ($CDS_coords && $mRNA_coords) { + #maybe created exon objects already + if ($self->{mRNA_exon_objs}) { + $self->trivial_refinement(); + } + return; + } + # intialize mRNA_exon_objs to array ref. + $self->{mRNA_exon_objs} = []; + #retrieve coordinate data. + my %mRNA = %$mRNA_coords; + my %CDS = %$CDS_coords; + my @mRNAcoords = keys %mRNA; + my @CDScoords = keys %CDS; + my (%new_mRNA, %new_CDS); + ## if correlation between mRNA exons and CDS exons, then map CDS's to mRNA's, otherwise, replicate CDSs as mRNAs + if ($#mRNAcoords >= $#CDScoords) { + + foreach my $mRNA_end5 (keys %mRNA) { + my $mRNA_end3 = $mRNA{$mRNA_end5}; + #find overlapping cds exon to mRNA exon + #easy to compare if in same orientation for all comparisons + my ($m1, $m2) = ($mRNA_end5 < $mRNA_end3) ? ($mRNA_end5, $mRNA_end3) : ($mRNA_end3, $mRNA_end5); + #create mRNA_exon_obj + my $mRNA_exon_obj = mRNA_exon_obj->new ($mRNA_end5, $mRNA_end3); + $new_mRNA{$mRNA_end5} = $mRNA_end3; + foreach my $CDS_end5 (keys %CDS) { + my $CDS_end3 = $CDS{$CDS_end5}; + my ($c1, $c2) = ($CDS_end5 < $CDS_end3) ? ($CDS_end5, $CDS_end3) : ($CDS_end3, $CDS_end5); + ## do overlap comparison; CDS must be contained within mRNA exon + if ( ($c1 >= $m1) && ($c2 <= $m2)) { + # found the contained CDS + $mRNA_exon_obj->{CDS_exon_obj} = CDS_exon_obj->new ($CDS_end5, $CDS_end3); + $new_CDS{$CDS_end5} = $CDS_end3; + last; + } + } + $self->add_mRNA_exon_obj($mRNA_exon_obj); + } + } else { # remap CDSs to mRNAS + print STDERR "ERROR: mRNA exons < CDS exons. Copying all CDS exons into mRNA exons. \n\n"; + foreach my $CDS_end5 (keys %CDS) { + my $CDS_end3 = $CDS{$CDS_end5}; + my $mRNA_exon_obj = mRNA_exon_obj->new ($CDS_end5, $CDS_end3); + $mRNA_exon_obj->{CDS_exon_obj} = CDS_exon_obj->new ($CDS_end5, $CDS_end3); + $self->add_mRNA_exon_obj($mRNA_exon_obj); + $new_mRNA{$CDS_end5} = $CDS_end3; + $new_CDS{$CDS_end5} = $CDS_end3; + } + } + + $self->trivial_refinement(); + + ## assign orientation to all children exon and CDS components. + my $strand = $self->get_orientation(); + foreach my $exon ($self->get_exons()) { + $exon->{strand} = $strand; + if (my $cds = $exon->get_CDS_exon_obj()) { + $cds->{strand} = $strand; + } + } + return; + +} + + +## alias +sub refine_gene_obj { + my $self = shift; + $self->refine_gene_object(); +} + + +=over 4 + +=item get_exons() + +BRetrieves a list of exons belonging to this Gene_obj + +B none + +B @exons + +@exons is an ordered list of mRNA_exon_obj; the first exon of the list corresponds to the first exon of the spliced gene. + +=back + +=cut + + ; + +sub get_exons { + my ($self) = shift; + if ($self->{mRNA_exon_objs} != 0) { + my @exons = (@{$self->{mRNA_exon_objs}}); + @exons = sort {$a->{end5}<=>$b->{end5}} @exons; + if ($self->{strand} eq '-') { + @exons = reverse (@exons); + } + return (@exons); + } else { + my @x = (); + return (@x); #empty array + } +} + + +## private +sub get_segments { + my $self = shift; + return ($self->get_exons()); +} + + + +=over 4 + +=item number_of_exons() + +B Provides the number of exons contained by the Gene + +B none + +B int + +=back + +=cut + + + +sub number_of_exons { + my $self = shift; + my $exon_number = $#{$self->{mRNA_exon_objs}} + 1; + return ($exon_number); +} + + + + + + + +=over 4 + +=item get_intron_coordinates() + +B Provides an ordered list of intron coordinates + +B none + +B ( [end5,end3], ....) + +A list of arrayRefs are returned providing the coordinates of introns, ordered from first intron to last intron within the gene. + +=back + +=cut + + ; + +sub get_intron_coordinates { + my $gene_obj = shift; + my $strand = $gene_obj->get_orientation(); + my @exons = $gene_obj->get_exons(); + ## exon list should already be sorted. + my @introns = (); + + my $num_exons = $#exons + 1; + if ($num_exons > 1) { #only genes with multiple exons will have introns. + if ($strand eq '+') { + my $first_exon = shift @exons; + while (@exons) { + my $next_exon = shift @exons; + my ($first_end5, $first_end3) = $first_exon->get_coords(); + my ($next_end5, $next_end3) = $next_exon->get_coords(); + my $intron_end5 = $first_end3 + 1; + my $intron_end3 = $next_end5 -1; + if ($intron_end5 < $intron_end3) { + push (@introns, [$intron_end5, $intron_end3]); + } + $first_exon = $next_exon; + } + } elsif ($strand eq '-') { + my $first_exon = shift @exons; + while (@exons) { + my $next_exon = shift @exons; + my ($first_end5, $first_end3) = $first_exon->get_coords(); + my ($next_end5, $next_end3) = $next_exon->get_coords(); + my $intron_end5 = $first_end3 - 1; + my $intron_end3 = $next_end5 +1; + if ($intron_end5 > $intron_end3) { + push (@introns, [$intron_end5, $intron_end3]); + } + $first_exon = $next_exon; + } + + } else { + die "Strand for gene_obj is not specified." . $gene_obj->toString(); + } + } + return (@introns); +} + + + + + +#private +sub trivial_refinement { + my $self = shift; + my @exons = $self->get_exons(); + $self->{num_exons} = scalar(@exons); + my (%mRNAexons, %CDSexons); + foreach my $exon (@exons) { + my ($exon_end5, $exon_end3) = $exon->get_mRNA_exon_end5_end3(); + $mRNAexons{$exon_end5} = $exon_end3; + my $cds; + if ($cds = $exon->get_CDS_obj()) { + my ($cds_end5, $cds_end3) = $cds->get_CDS_end5_end3(); + $CDSexons{$cds_end5} = $cds_end3; + } + } + my @mRNAexonsEnd5s = sort {$a<=>$b} keys %mRNAexons; + my @CDSexonsEnd5s = sort {$a<=>$b} keys %CDSexons; + my $strand = 0; #initialize. + foreach my $mRNAend5 (@mRNAexonsEnd5s) { + my $mRNAend3 = $mRNAexons{$mRNAend5}; + if ($mRNAend5 == $mRNAend3) {next;} + $strand = ($mRNAend5 < $mRNAend3) ? '+':'-'; + last; + } + $self->{strand} = $strand; + + ## determine gene and model boundaries: + my ($gene_end5, $gene_end3, $model_end5, $model_end3); + my @gene_coords = sort {$a<=>$b} %mRNAexons; + my @model_coords = sort {$a<=>$b} %CDSexons; + my $gene_lend = shift @gene_coords; + my $gene_rend = pop @gene_coords; + ## bound gene by transcript span + ($gene_end5, $gene_end3) = ($strand eq "+") ? ($gene_lend, $gene_rend) : ($gene_rend, $gene_lend); + if (@model_coords) { + ## bound model by protein coding span + my $model_lend = shift @model_coords; + my $model_rend = pop @model_coords; + ($model_end5, $model_end3) = ($strand eq "+") ? ($model_lend, $model_rend) : ($model_rend, $model_lend); + } else { + ## give it gene boundaries instead: + ($model_end5, $model_end3) = ($gene_end5, $gene_end3); + } + + $self->{gene_span} = [$gene_end5, $gene_end3]; + $self->{gene_length} = abs ($gene_end3 - $gene_end5) + 1; + $self->{mid_pt} = int (($gene_end5 + $gene_end3)/2); + $self->{model_span} = [$model_end5, $model_end3]; + + ## Refine isoforms if they exist. + if (my @isoforms = $self->get_additional_isoforms()) { + my @gene_span_coords = $self->get_gene_span(); + foreach my $isoform (@isoforms) { + $isoform->refine_gene_object(); + push (@gene_span_coords, $isoform->get_gene_span()); + } + @gene_span_coords = sort {$a<=>$b} @gene_span_coords; + my $lend = shift @gene_span_coords; + my $rend = pop @gene_span_coords; + my $strand = $self->{strand}; + if ($strand eq '-') { + ($lend, $rend) = ($rend, $lend); + } + my $gene_length = abs ($lend -$rend) + 1; + foreach my $gene ($self, @isoforms) { + $gene->{gene_span} = [$lend, $rend]; + $gene->{gene_length} = $gene_length; + } + } + +} + + + + +=over 4 + +=item add_mRNA_exon_obj() + +B Used to add a single mRNA_exon_obj to the Gene_obj + +B mRNA_exon_obj + +B none + +=back + +=cut + + ; + +sub add_mRNA_exon_obj { + my ($self) = shift; + my ($mRNA_exon_obj) = shift; + if (!ref($self->{mRNA_exon_objs})) { + $self->{mRNA_exon_objs} = []; + } + my $index = $#{$self->{mRNA_exon_objs}}; + $index++; + $self->{mRNA_exon_objs}->[$index] = $mRNA_exon_obj; +} + +#private +## forcibly set protein sequence value + + +sub set_protein_sequence { + my $self = shift; + my $protein = shift; + if ($protein) { + $self->{protein_seq} = $protein; + $self->{protein_seq_length} = length ($protein); + } else { + print STDERR "No incoming protein sequence to set to.\n" . $self->toString(); + } +} + +#private +## forcibly set CDS sequence value +sub set_CDS_sequence { + my $self = shift; + my $cds_seq = shift; + if ($cds_seq) { + $self->{CDS_sequence} = $cds_seq; + $self->{CDS_sequence_length} = length ($cds_seq); + } else { + print STDERR "No incoming CDS sequence to set to\n" . $self->toString(); + } +} + +#private +sub set_cDNA_sequence { + my $self = shift; + my $cDNA_seq = shift; + if ($cDNA_seq) { + $self->{cDNA_sequence} = $cDNA_seq; + $self->{cDNA_sequence_length} = length($cDNA_seq); + } else { + print STDERR "No incoming cDNA sequence to set to.\n" . $self->toString(); + } +} + +#private +sub set_gene_sequence { + my $self = shift; + my $seq = shift; + if ($seq) { + $self->{gene_sequence} = $seq; + $self->{gene_sequence_length} = length ($seq); + } else { + print STDERR "No incoming gene sequence to set to\n" . $self->toString(); + } +} + + +=over 4 + +=item create_all_sequence_types() + +B Given a scalar reference to the genomic sequence, the CDS, cDNA, unspliced transcript and protein sequences are constructed and populated within the Gene_obj + +B $genomic_seq_ref, [%params] + +B 0|1 + +returns 1 upon success, 0 upon failure + +By default, the protein and CDS sequence are populated. If you want the unspliced genomic sequence, you need to specify this in the attributes: + + %params = ( potein => 1, + CDS => 1, + cDNA => 1, + unspliced_transcript => 0) + + +=back + +=cut + + +## Create all gene sequences (protein, cds, cdna, genomic) +sub create_all_sequence_types { + my $self = shift; + my $big_seq_ref = shift; + my %atts = @_; + + unless (ref($big_seq_ref) eq 'SCALAR') { + print STDERR "I require a sequence reference to create sequence types\n"; + return (undef()); + } + $self->create_cDNA_sequence($big_seq_ref) unless (exists($atts{cDNA}) && $atts{cDNA}); + $self->create_gene_sequence($big_seq_ref, 1) if ($atts{unspliced_transcript}); #highlight exons by default. + + if ($self->is_coding_gene()) { + $self->create_CDS_sequence ($big_seq_ref) unless (exists ($atts{CDS}) && $atts{CDS}); + $self->create_protein_sequence($big_seq_ref) unless (exists ($atts{protein}) && $atts{protein}); + } + + if (my @isoforms = $self->get_additional_isoforms()) { + foreach my $isoform (@isoforms) { + $isoform->create_all_sequence_types($big_seq_ref, %atts); + } + } + return(1); +} + +#private +## Create cDNA sequence +sub create_cDNA_sequence { + my $self = shift; + my $seq_ref = shift; + my $sequence_ref; + unless ($seq_ref) { + print STDERR "The parent sequence must be specified for the cDNA creation method\n"; + return; + } + ## hopefully the sequence came in as a reference. If not, make one to it. + ## Don't want to pass chromosome sequences in by value! + if (ref($seq_ref)) { + $sequence_ref = $seq_ref; + } else { + $sequence_ref = \$seq_ref; + } + my @exons = $self->get_exons(); + my $strand = $self->{strand}; + my $cDNA_seq = ""; + foreach my $exon_obj (sort {$a->{end5}<=>$b->{end5}} @exons) { + my $c1 = $exon_obj->{end5}; + my $c2 = $exon_obj->{end3}; + ## sequence retrieval coordinates must be in forward orientation + my ($coord1, $coord2) = ($strand eq '+') ? ($c1, $c2) : ($c2, $c1); + $cDNA_seq .= substr ($$sequence_ref, ($coord1 - 1), ($coord2 - $coord1 + 1)); + } + if ($strand eq '-') { + $cDNA_seq = &reverse_complement($cDNA_seq); + } + $self->set_cDNA_sequence($cDNA_seq); + return ($cDNA_seq); +} + +#private +## create a CDS sequence, and populate the protein field. +sub create_CDS_sequence { + my $self = shift; + my $seq_ref = shift; + my $sequence_ref; + unless ($seq_ref) { + print STDERR "The parent sequence must be specified for the CDS creation method\n"; + return; + } + + unless ($self->is_coding_gene()) { + print STDERR "Warning: No coding region specified for gene: " . $self->toString(); + return(""); + } + + + ## hopefully the sequence came in as a reference. If not, make one to it. + ## Don't want to pass chromosome sequences in by value! + if (ref($seq_ref)) { + $sequence_ref = $seq_ref; + } else { + $sequence_ref = \$seq_ref; + } + my @exons = $self->get_exons(); + my $strand = $self->{strand}; + my $cds_seq = ""; + foreach my $exon_obj (sort {$a->{end5}<=>$b->{end5}} @exons) { + my $CDS_obj = $exon_obj->get_CDS_obj(); + if (ref $CDS_obj) { + my ($c1, $c2) = $CDS_obj->get_CDS_end5_end3(); + ## sequence retrieval coordinates must be in forward orientation + my ($coord1, $coord2) = ($strand eq '+') ? ($c1, $c2) : ($c2, $c1); + $cds_seq .= substr ($$sequence_ref, ($coord1 - 1), ($coord2 - $coord1 + 1)); + } + } + if ($strand eq '-') { + $cds_seq = &reverse_complement($cds_seq); + } + $self->set_CDS_sequence($cds_seq); + + return ($cds_seq); +} + + + +sub is_coding_gene { + my $self = shift; + + if ($self->get_CDS_length()) { + return(1); + } + else { + return(0); + } +} + + + +#private +## Translation requires parent nucleotide sequence (bac, chromosome, whatever). +sub create_protein_sequence { + my $self = shift; + my $seq_ref = shift; # optional + + unless ($self->is_coding_gene()) { + print STDERR "Warning: No coding sequence for gene: " . $self->toString(); + return(""); + } + + my $cds_sequence = $self->get_CDS_sequence(); + unless ($cds_sequence) { + + ## if has a CDS, then try to translate it if the genome sequence is available. + + unless (ref($seq_ref) eq 'SCALAR') { + print STDERR "I require an assembly sequence ref if the CDS is unavailable\n"; + return; + } + $cds_sequence = $self->create_CDS_sequence($seq_ref); + } + my $protein = &Nuc_translator::get_protein ($cds_sequence); + $self->set_protein_sequence($protein); + return ($protein); +} + +#private +## Create the unspliced nucleotide transcript +sub create_gene_sequence { + my $self = shift; + my $big_seq_ref = shift; + my $highlight_exons_flag = shift; #upcases exons, lowcases introns. + unless (ref ($big_seq_ref) eq 'SCALAR') { + print STDERR "I require a reference to the assembly sequence!!\n"; + return (undef()); + } + my $strand = $self->{strand}; + my ($gene_seq); + if ($highlight_exons_flag) { + my @exons = sort {$a->{end5}<=>$b->{end5}} $self->get_exons(); + my $exon = shift @exons; + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + $gene_seq = uc (substr ($$big_seq_ref, $lend - 1, $rend - $lend + 1)); + my $prev_rend = $rend; + while (@exons) { + $exon = shift @exons; + ## Add intron, then exon + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + $gene_seq .= lc (substr ($$big_seq_ref, $prev_rend, $lend - $prev_rend-1)); + $gene_seq .= uc (substr ($$big_seq_ref, $lend - 1, $rend - $lend + 1)); + $prev_rend = $rend; + } + + } else { #just get the sequence spanned by min and max coords + my ($coord1, $coord2) = sort {$a<=>$b} $self->get_gene_span(); + $gene_seq = substr ($$big_seq_ref, ($coord1 - 1), ($coord2 - $coord1 + 1)); + } + + $gene_seq = &reverse_complement($gene_seq) if ($strand eq '-'); + $self->set_gene_sequence($gene_seq); + return ($gene_seq); +} + +## retrieving the sequences + +=over 4 + +=item get_protein_sequence() + +B Retrieves the protein sequence + +B none + +B $protein + +Note: You must have called create_all_sequence_types($genomic_ref) before protein sequence is available for retrieval. + + +=back + +=cut + + ; + +sub get_protein_sequence { + my $self = shift; + return ($self->{protein_seq}); +} + +## alias +sub get_protein_seq { + my $self = shift; + return ($self->get_protein_sequence()); +} + + + +=over 4 + +=item get_CDS_sequence() + +B Retrieves the CDS sequence. The CDS sequence is the protein-coding nucleotide sequence. + +B none + +B $cds + +Note: You must have called create_all_sequence_types($genomic_ref) before protein sequence is available for retrieval. + +=back + +=cut + + +sub get_CDS_sequence { + my $self = shift; + return ($self->{CDS_sequence}); +} + +=over 4 + +=item get_cDNA_sequence() + +B Retrieves the tentative cDNA sequence for the Gene. The cDNA includes the CDS with potential UTR extensions. + +B none + +B $cdna + +Note: You must have called create_all_sequence_types($genomic_ref) before protein sequence is available for retrieval. + + +=back + +=cut + + + +sub get_cDNA_sequence { + my $self = shift; + return ($self->{cDNA_sequence}); +} + + + +sub get_CDS_length { + my $self = shift; + + my $cds_length = 0; + + my @exons = $self->get_exons(); + foreach my $exon (@exons) { + if (my $cds = $exon->get_CDS_obj()) { + $cds_length += $cds->length(); + } + } + + + return ($cds_length); +} + +sub get_cDNA_length { + my $self = shift; + + my $cdna_length = 0; + + my @exons = $self->get_exons(); + foreach my $exon (@exons) { + $cdna_length += $exon->length(); + } + + return($cdna_length); + +} + + + + + + + +=over 4 + +=item get_gene_sequence() + +B Retrieves the unspliced transcript of the gene. + +B none + +B $unspliced_transcript + +=back + +=cut + + +sub get_gene_sequence { + my $self = shift; + return ($self->{gene_sequence}); +} + + + + +=over 4 + +=item get_gene_synonyms() + +B Retrieves the Model_feat_name(s) for the synonomous gene models found on other BACs or contigs. + +B none + +B @model_feat_names + + +For Arabidopsis, gene models are found within overlapping regions of BAC sequences, in which the gene models are annotated on both corresponding BACs. Given a Gene_obj for a model on one BAC, the synomous gene on the overlapping BAC can be identified via this method. + + +=back + +=cut + + +sub get_gene_synonyms { + my $self = shift; + return (@{$self->{gene_synonyms}}); +} + + + +=over 4 + +=item clear_sequence_info() + +B Clears the sequence fields stored within a Gene_obj, including the CDS, cDNA, gene_sequence, and protein sequence. Often, these sequence fields, when populated, can consume large amounts of memory in comparison to the coordinate and functional annotation data. This method is useful to clear this memory when the sequences are not needed. The create_all_sequence_types($genomic_seq_ref) can be called again later to repopulate these sequences when they are needed. + +B none + +B none + +=back + +=cut + + +## sequences consume huge amounts of memory in comparison to other gene features. +## want to clear them from time to time to save memory. + + ; + +sub clear_sequence_info { + my $self = shift; + $self->{protein_seq} = undef; + $self->{CDS_sequence} = undef; + $self->{cDNA_sequence} = undef; + $self->{gene_sequence} = undef; +} + + +=over 4 + +=item set_gene_type() + +B Sets the type of gene. Expected types include: + + protein-coding #default setting + rRNA + snoRNA + snRNA + tRNA + + ...or others as needed. Nothing is restricted. + +B $type + +B none + +=back + +=cut + + +#### +sub set_gene_type { + my ($self) = shift; + my ($gene_type) = shift; + $self->{gene_type} = $gene_type; +} + + +=over 4 + +=item adjust_gene_coordinates() + +B Used to add or subtract a specified number of bases from each gene component coordinate. + +B $adj_amount + +$adj_amoount is a positive or negative integer. + +B none + +=back + +=cut + + + ; + +#### +# add value to all gene component coordinates +sub adjust_gene_coordinates { + my $self = shift; + my $adj_amount = shift; + my @exons = $self->get_exons(); + foreach my $exon (@exons) { + my ($end5, $end3) = $exon->get_coords(); + $exon->set_coords($end5 + $adj_amount, $end3 + $adj_amount); + my $cds = $exon->get_CDS_obj(); + if (ref $cds) { + my ($end5, $end3) = $cds->get_coords(); + $cds->set_coords($end5 + $adj_amount, $end3 + $adj_amount); + } + } + + ## don't forget about alt splicing isoforms! + my @isoforms = $self->get_additional_isoforms(); + foreach my $isoform (@isoforms) { + $isoform->adjust_gene_coordinates($adj_amount); + } + $self->refine_gene_object(); +} + + + + +=over 4 + +=item toString() + +B Textually describes the Gene_obj including coordinates and attributes. + +B <%attributes_list> + +%attributes_list is optional and can control whether certain attributes are included in the textual output + +Default settings are: + + %attributes_list = ( + -showIsoforms => 1, #set to 0 to avoid isoform info to the text output. + -showSeqs => 0 #set to 1 for avoiding protein, cds, genomic, cdna seqs as output. + ) + +B $text + +=back + +=cut + + ; + + +## retrieve text output describing the gene. +sub toString { + my $self = shift; + my %atts = @_; + # atts defaults: + # -showIsoforms=>1 + # -showSeqs => 0 + + my $output = ""; + foreach my $key (keys %$self) { + my $value = $self->{$key}; + unless (defined $value) { next;} + if (ref $value) { + if ($key =~ /secondary/ && ref $value eq "ARRAY") { + foreach my $val (@$value) { + $output .= "\t\t$key\t$val\n"; + } + } + + + } else { + if ($self->{is_pseudogene} && $key =~ /cds|cdna|protein/i && $key =~ /seq/) { + next; + } + if ((!$atts{-showSeqs}) && $key =~/seq/) { next; } + if ( ($value eq '0' || !defined($value)) && $key !~/^is_/) { next;} #dont print unpopulated info. + $output .= "\t$key:\t$value\n"; + } + } + $output .= "\tgene_synonyms: @{$self->{gene_synonyms}}\n"; + + $output .= "\tmRNA_coords\t"; + + if (ref ($self->{mRNA_coords}) eq "HASH") { + foreach my $end5 (sort {$a<=>$b} keys %{$self->{mRNA_coords}}) { + $output .= "$end5-$self->{mRNA_coords}->{$end5} "; + } + } + $output .= "\n" + . "\tCDS_coords\t"; + if (ref ($self->{CDS_coords}) eq "HASH") { + foreach my $end5 (sort {$a<=>$b} keys %{$self->{CDS_coords}}) { + $output .= "$end5-$self->{CDS_coords}->{$end5} "; + } + } + + my @exons = $self->get_exons(); + foreach my $exon (@exons) { + $output .= "\n\t\tRNA-exon: $exon->{end5}, $exon->{end3}\t"; + my $cds = $exon->{CDS_exon_obj}; + if ($cds) { + $output .= "CDS-exon: $cds->{end5}, $cds->{end3}"; + } + } + + if (ref $self->{gene_span}) { + my ($gene_end5, $gene_end3) = @{$self->{gene_span}}; + $output .= "\n\tgene_span: $gene_end5-$gene_end3"; + } + if (ref $self->{model_span}) { + my ($model_end5, $model_end3) = @{$self->{model_span}}; + $output .= "\n\tmodel_span: $model_end5-$model_end3"; + } + my @gene_ontology_objs = $self->get_gene_ontology_objs(); + if (@gene_ontology_objs) { + $output .= "\n\tGene Ontology Assignments:\n"; + foreach my $go_assignment (@gene_ontology_objs) { + $output .= "\t" . $go_assignment->toString(); + } + } + + unless (defined ($atts{-showIsoforms}) && $atts{-showIsoforms} == 0) { + foreach my $isoform ($self->get_additional_isoforms()) { + $output .= "\n\n\tISOFORM:\n" . $isoform->toString(); + } + } + $output .= "\n\n"; #spacer at terminus + return ($output); +} + + +#### +## Splice site validation section +#### + +=over 4 + +=item validate_splice_sites() + +B Validates the presence of consensus splice sites + +B $genomic_seq_ref + +$genomic_seq_ref is a scalar reference to the string containing the genomic sequence. + +B $errors + +If the empty string ("") is returned, then no inconsistencies were identified. + +=back + +=cut + + ; + +#### +sub validate_splice_sites { + my $self = shift; + my $asmbl_seq_ref = shift; + unless (ref ($asmbl_seq_ref)) { + print STDERR "I require a sequence reference\n"; + return (undef()); + } + my $error_string = ""; + my $strand = $self->{strand}; + my @exons = $self->get_exons(); + my $num_exons = $#exons + 1; + if ($num_exons == 1) { + #no splice sites to confirm. + return (""); + } + for (my $i = 1; $i <= $num_exons; $i++) { + my $exon_type; + if ($i == 1) { + $exon_type = "initial"; + } elsif ($i == $num_exons) { + $exon_type = "terminal"; + } else { + $exon_type = "internal"; + } + my $exon = $exons[$i - 1]; + my ($exon_end5, $exon_end3) = $exon->get_mRNA_exon_end5_end3(); + my ($coord1, $coord2) = sort {$a<=>$b} ($exon_end5, $exon_end3); + ## get two coordinate sets corresponding to potential splice sites + my $splice_1_start = $coord1-2-1; + my $splice_2_start = $coord2-1+1; + #print "confirming splice sites at " . ($splice_1_start +1) . " and " . ($splice_2_start + 1) . "\n"if $SEE; + my $splice_1 = substr ($$asmbl_seq_ref, $splice_1_start, 2); + my $splice_2 = substr ($$asmbl_seq_ref, $splice_2_start, 2); + my ($acceptor, $donor) = ($strand eq '+') ? ($splice_1, $splice_2) : (&reverse_complement($splice_2), &reverse_complement($splice_1)); + my $check_acceptor = ($acceptor =~ /ag/i); + my $check_donor = ($donor =~ /gt|gc/i); + ## associate results of checks with exon type. + if ($exon_type eq "initial" || $exon_type eq "internal") { + unless ($check_donor) { + $error_string .= "non-consensus $donor donor splice site at $coord1\n"; + } + } + + if ($exon_type eq "internal" || $exon_type eq "terminal") { + unless ($check_acceptor) { + $error_string .= "\tnon-consensus $acceptor acceptor splice site at $coord2\n"; + } + } + } + return ($error_string); +} + + + +=over 4 + +=item get_annot_text() + +B Provides basic functional annotation for a Gene_obj + +B none + +B $string + +$string includes locus, pub_locus, com_name, and pub_comment + +=back + +=cut + + + ; + +#### +sub get_annot_text { + my $self = shift; + my $locus = $self->{locus}; + my $pub_locus = $self->{pub_locus}; + my $com_name = $self->{com_name}; + my $pub_comment = $self->{pub_comment}; + my $text = ""; + foreach my $token ($locus, $pub_locus, $com_name, $pub_comment) { + if ($token) { + $text .= "$token "; + } + } + return ($text); +} + + + +=over 4 + +=item add_isoform() + +B Adds a Gene_obj to an existing Gene_obj as an alternative splicing variant. + +B Gene_obj + +B none + +=back + +=cut + + ; +sub add_isoform { + my $self = shift; + my @gene_objs = @_; + foreach my $gene_obj (@gene_objs) { + $self->{num_additional_isoforms}++; + push (@{$self->{additional_isoforms}}, $gene_obj); + } +} + + + + + +=over 4 + +=item has_additional_isoforms() + +B Provides number of additional isoforms. Typically used as a boolean. + +B none + +B number of additional isoforms (int) + +If no additional isoforms exist, returns 0 + + +boolean usage: + +0 = false (has no more) +nonzero = true (has additional isoforms) + +=back + +=cut + +sub has_additional_isoforms { + my $self = shift; + return ($self->{num_additional_isoforms}); +} + + + +=over 4 + +=item delete_isoforms() + +B removes isoforms stored in this Gene_obj (assigning to a new anonymous arrayref) + +B Gene_obj + +B none + +=back + +=cut + +sub delete_isoforms { + my $self = shift; + $self->{num_additional_isoforms} = 0; + $self->{additional_isoforms} = []; +} + + + + + +=over 4 + +=item get_additional_isoforms() + +B Retrieves the additional isoforms for a given Gene_obj + +B none + +B @Gene_objs + +If no additional isoforms exist, an empty array is returned. + +=back + +=cut + + +sub get_additional_isoforms { + my $self = shift; + return (@{$self->{additional_isoforms}}); +} + + + +=over 4 + +=item get_orientation() + +B Retrieves the strand orientation of the Gene_obj + +B none + +B +|- + +=back + +=cut + + +sub get_orientation { + my $self = shift; + return ($self->{strand}); +} + + + +sub get_strand { ## preferred + my $self = shift; + return($self->get_orientation()); +} + + + +=over 4 + +=item add_gene_ontology_objs() + +B Adds a list of Gene_ontology objects to a Gene_obj + +B @Gene_ontology_objs + +@Gene_ontology_objs is a list of objects instantiated from Gene_ontology.pm + +B none + +=back + +=cut + + +sub add_gene_ontology_objs { + my ($self, @ontology_objs) = @_; + push (@{$self->{GeneOntology}}, @ontology_objs); +} + + + +=over 4 + +=item get_gene_ontology_objs() + +B Retrieves Gene_ontology objs assigned to the Gene_obj + +B none + +B @Gene_ontology_objs + +@Gene_ontology_objs are objects instantiated from package Gene_ontology (See Gene_ontology.pm) + +=back + +=cut + + ; + +sub get_gene_ontology_objs { + my $self = shift; + if (ref ($self->{GeneOntology})) { + return (@{$self->{GeneOntology}}); + } else { + return (()); + } +} + + +=over 4 + +=item set_5prime_partial() + +B Sets the status of the is_5prime_partial attribute + +B 1|0 + +B none + + +5prime partials are partial on their 5prime end and lack start codons. + + +=back + +=cut + +sub set_5prime_partial() { + my $self = shift; + my $value = shift; + $self->{is_5prime_partial} = $value; +} + + + +=over 4 + +=item set_3prime_partial() + +B Sets the is_3prime_partial status + +B 1|0 + +B none + +3prime partials are partial on their 3prime end and lack stop codons. + +=back + +=cut + + +sub set_3prime_partial() { + my $self = shift; + my $value = shift; + $self->{is_3prime_partial} = $value; +} + + + +=over 4 + +=item is_5prime_partial() + +B Retrieves the 5-prime partial status of the gene. + +B none + +B 1|0 + +=back + +=cut + + +sub is_5prime_partial() { + my $self = shift; + return ($self->{is_5prime_partial}); +} + + +=over 4 + +=item is_3prime_partial() + +B Retrieves the 3-prime partial status of the gene. + +B none + +B 1|0 + +=back + +=cut + + +sub is_3prime_partial() { + my $self = shift; + return ($self->{is_3prime_partial}); +} + +=over 4 + +=item get_5prime_UTR_coords + + +B returns a list of coordinate pairs corresponding to the 5\' UTR coordinates + +B none + +B ([end5,end3], ...) or empty list if none exist + +=back + +=cut + + + ; + +sub get_5prime_UTR_coords { + my $self = shift; + + my $strand = $self->get_orientation(); + + my @exons = $self->get_exons(); + + my $seen_CDS_flag = 0; + + my @utr_coords; + foreach my $exon (@exons) { #relying on a sorted list + my ($exon_end5, $exon_end3) = $exon->get_coords(); + if (my $cds = $exon->get_CDS_obj()) { + my ($cds_end5, $cds_end3) = $cds->get_coords(); + if ($exon_end5 != $cds_end5) { + my $adj_utr_end3_coord = ($strand eq '+') ? ($cds_end5 -1) : ($cds_end5 +1); + push (@utr_coords, [$exon_end5, $adj_utr_end3_coord]); + } + + $seen_CDS_flag = 1; + + } else { + push (@utr_coords, [$exon_end5, $exon_end3]); + } + + if ($seen_CDS_flag) { + last; + } + + } + + return (@utr_coords); +} + + + +=over 4 + +=item get_3prime_UTR_coords + + +B returns a list of coordinate pairs corresponding to the 3\' UTR coordinates + +B none + +B ([end5,end3], ...) or empty list if none exist + +=back + +=cut + + ; + +sub get_3prime_UTR_coords { + my $self = shift; + + my $strand = $self->get_orientation(); + + my @exons = reverse $self->get_exons(); + + my @utr_coords; + my $seen_CDS_flag = 0; + foreach my $exon (@exons) { #relying on a reverse sorted list (3' exons should come first) + my ($exon_end5, $exon_end3) = $exon->get_coords(); + if (my $cds = $exon->get_CDS_obj()) { + $seen_CDS_flag = 1; + my ($cds_end5, $cds_end3) = $cds->get_coords(); + if ($exon_end3 != $cds_end3) { + my $adj_utr_end5_coord = ($strand eq '+') ? ($cds_end3 +1) : ($cds_end3 -1); + push (@utr_coords, [$adj_utr_end5_coord, $exon_end3]); + } + + } else { + push (@utr_coords, [$exon_end5, $exon_end3]); + } + if ($seen_CDS_flag) { + last; + } + } + + if (@utr_coords) { + @utr_coords = reverse @utr_coords; + } + + return (@utr_coords); +} + + + + + +=over 4 + +=item has_UTRs() + +B indicates presence of UTR annotated in Gene + +B none + +B ( has_5prime_UTR() || has_3prime_UTR() ) + +=back + +=cut + +sub has_UTRs { + my $self = shift; + return ( ($self->has_5prime_UTR() || $self->has_3prime_UTR() ) ); +} + + + +#### +sub has_5prime_UTR { + my $self = shift; + return ($self->get_5prime_UTR_length() > 2); +} + +#### +sub has_3prime_UTR { + my $self = shift; + return($self->get_3prime_UTR_length() > 2); +} + + +#### +sub get_5prime_UTR_length { + my $self = shift; + my @prime5_UTR_coords = $self->get_5prime_UTR_coords(); + + my $len = 0; + for my $coordset (@prime5_UTR_coords) { + my ($lend, $rend) = sort {$a<=>$b} @$coordset; + $len += ($rend - $lend) + 1; + } + return($len); +} + + +#### +sub get_3prime_UTR_length { + my $self = shift; + my @prime3_UTR_coords = $self->get_3prime_UTR_coords(); + + my $len = 0; + for my $coordset (@prime3_UTR_coords) { + my ($lend, $rend) = sort {$a<=>$b} @$coordset; + $len += ($rend - $lend) + 1; + } + return($len); +} + + + +=over 4 + +=item get_5prime_UTR_sequence() + +B retrieves 5prime UTR sequence + +B genome sequence reference + +B string + +=back + +=cut + +#### +sub get_5prime_UTR_sequence { + my $self = shift; + my ($genome_seq_ref) = @_; + unless (ref $genome_seq_ref eq "SCALAR") { + confess "error, require genome sequence string reference"; + } + + unless ($self->has_5prime_UTR()) { + return ""; + } + + my $orientation = $self->get_orientation(); + my @coords = $self->get_5prime_UTR_coords(); + + @coords = sort {$a->[0]<=>$b->[0]} @coords; + + my $UTR_seq = ""; + foreach my $coordset (@coords) { + my ($lend, $rend) = sort {$a<=>$b} @$coordset; + + my $length = $rend - $lend + 1; + $UTR_seq .= substr($$genome_seq_ref, $lend - 1, $length); + } + + if ($orientation eq '-') { + $UTR_seq = &reverse_complement($UTR_seq); + } + + ## verify: + $self->create_all_sequence_types($genome_seq_ref); + my $cDNA = $self->get_cDNA_sequence(); + + + unless (index($cDNA, $UTR_seq) == 0) { + confess "Error, couldn't find UTR in cDNA"; + } + + + return ($UTR_seq); +} + + +=over 4 + +=item get_3prime_UTR_sequence() + +B retrieves 5prime UTR sequence + +B genome sequence reference + +B string + +=back + +=cut + +#### +sub get_3prime_UTR_sequence { + my $self = shift; + my ($genome_seq_ref) = @_; + unless (ref $genome_seq_ref eq "SCALAR") { + confess "error, require genome sequence string reference"; + } + + unless ($self->has_3prime_UTR()) { + return ""; + } + + my $orientation = $self->get_orientation(); + my @coords = $self->get_3prime_UTR_coords(); + + @coords = sort {$a->[0]<=>$b->[0]} @coords; + + my $UTR_seq = ""; + foreach my $coordset (@coords) { + my ($lend, $rend) = sort {$a<=>$b} @$coordset; + + my $length = $rend - $lend + 1; + $UTR_seq .= substr($$genome_seq_ref, $lend - 1, $length); + } + + if ($orientation eq '-') { + $UTR_seq = &reverse_complement($UTR_seq); + } + + ## verify: + $self->create_all_sequence_types($genome_seq_ref); + my $cDNA = $self->get_cDNA_sequence(); + my $cDNA_length = length($cDNA); + my $utr_length = length($UTR_seq); + + my $utr_start_pos = $cDNA_length - $utr_length + 1; + + unless ((my $cDNA_utr = lc substr($cDNA, $utr_start_pos - 1, $utr_length)) eq lc $UTR_seq) { + confess "Error, 3' UTR extracted from cDNA is different from UTR sequence extracted from genome.\n" + . "cDNA_utr:\n$cDNA_utr\nUTR_from_genome:\n$UTR_seq\n\n"; + } + + + return ($UTR_seq); +} + + + + +=over 4 + +=item trim_UTRs() + +B Trims the UTR of the Gene_obj so that the Exon coordinates are identical to the CDS coordinates. Exons which lack CDS components and are completely UTR are removed. + +B none + +B none + +=back + +=cut + + ; + +sub trim_UTRs { + my $self = shift; + + ## adjust exon coordinates to CDS coordinates. + ## if cds doesn't exist, rid exon: + + my @new_exons; + + my @exons = $self->get_exons(); + foreach my $exon (@exons) { + if (my $cds = $exon->get_CDS_obj()) { + my ($exon_end5, $exon_end3) = $exon->get_coords(); + my ($cds_end5, $cds_end3) = $cds->get_coords(); + + if ($exon_end5 != $cds_end5 || $exon_end3 != $cds_end3) { + $exon->set_coords($cds_end5, $cds_end3); + } + push (@new_exons, $exon); + } + } + $self->{mRNA_exon_objs} = 0; #clear current gene structure + $self->{mRNA_exon_objs} = \@new_exons; #replace gene structure + $self->refine_gene_object(); #update + return ($self); +} + + + + +=over 4 + +=item remove_CDS_exon() + +B Removes any existing CDS_exon_obj from this mRNA_exon_obj + +B none + +B none + +=back + +=cut + +sub remove_CDS_exon { + my $self = shift; + $self->{CDS_exon_obj} = 0; +} + + + + + +=over 4 + +=item get_gene_names() + +B Retrieves gene names (primary gene name followed by secondary gene names, "$;" delimited. + +B none + +B string + + see $gene_obj->{gene_name} + see $gene_obj->get_secondary_names() + +secondary gene names sorted lexicographically + + +=back + +=cut + + + + +#### +sub get_gene_names { + my $gene_obj = shift; + my @gene_names; + if ($gene_obj->{gene_name}) { + push (@gene_names, $gene_obj->{gene_name}); + } + if (my @secondary_names = $gene_obj->get_secondary_gene_names()) { + push (@gene_names, @secondary_names); + } + my $ret_gene_names = join ("$;" , @gene_names); + return ($ret_gene_names); +} + + + +=over 4 + +=item get_secondary_gene_names() + +B Retrieves secondary gene names as a "$;" delimited string. + +B none + +B string + +=back + +=cut + + +#### +sub get_secondary_gene_names { + my ($gene_obj) = @_; + return (sort @{$gene_obj->{secondary_gene_names}}); +} + + + + +=over 4 + +=item get_product_names() + +B Retrieves product name, with the primary product name followed by secondary product names, delimited by "$;" + +B none + +B string + + see $gene_obj->{com_name} for primary product name + see $gene_obj->get_secondary_product_names() + +=back + +=cut + + ; + +#### +sub get_product_names { + my $gene_obj = shift; + my @product_names; + if ($gene_obj->{com_name}) { + push (@product_names, $gene_obj->{com_name}); + } + if (my @secondary_names = $gene_obj->get_secondary_product_names()) { + push (@product_names, @secondary_names); + } + my $ret_product_names = join ("$;", @product_names); + return ($ret_product_names); +} + + + +=over 4 + +=item get_secondary_product_names() + +B Retrieves secondary product names, delimited by "$;" and sorted lexicographically. + +B none + +B string + +=back + +=cut + + +#### +sub get_secondary_product_names { + my ($gene_obj) = @_; + return (sort @{$gene_obj->{secondary_product_names}}); +} + + + +=over 4 + +=item get_gene_symbols() + +B Retrieves primary gene symbol followed by secondary gene symbols, delimited by "$;" + +B none + +B string + + see $gene_obj->{gene_sym} + see $gene_obj->get_secondary_gene_symbols() + +=back + +=cut + + ; + +#### +sub get_gene_symbols { + my $gene_obj = shift; + my @gene_symbols; + if ($gene_obj->{gene_sym}) { + push (@gene_symbols, $gene_obj->{gene_sym}); + } + if (my @secondary_symbols = $gene_obj->get_secondary_gene_symbols()) { + push (@gene_symbols, @secondary_symbols); + } + my $ret_gene_symbols = join ("$;", @gene_symbols); + return ($ret_gene_symbols); +} + + +=over 4 + +=item get_secondary_gene_symbols() + +B Retrieves secondary gene symbols, delimited by "$;" and sorted lexicographically + +B none + +B string + +=back + +=cut + + +#### +sub get_secondary_gene_symbols { + my ($gene_obj) = @_; + return (sort @{$gene_obj->{secondary_gene_symbols}}); +} + + + +=over 4 + +=item get_ec_numbers() + +B Retrieves primary EC number followed by secondary EC numbers, "$;" delimited + +B none + +B string + + see $gene_obj->{ec_num} + see $gene_obj->get_secondary_ec_numbers() + +=back + +=cut + + ; + +#### +sub get_ec_numbers { + my $gene_obj = shift; + my @ec_numbers; + if ($gene_obj->{ec_num}) { + push (@ec_numbers, $gene_obj->{ec_num}); + } + if (my @secondary_ec_numbers = $gene_obj->get_secondary_ec_numbers()) { + push (@ec_numbers, @secondary_ec_numbers); + } + my $ret_ec_numbers = join ("$;", @ec_numbers); + return ($ret_ec_numbers); +} + + + +=over 4 + +=item get_secondary_ec_numbers() + +B Retrieves secondary EC numbers, "$;" delimited and sorted lexicographically + +B none + +B string + + +=back + +=cut + + +#### +sub get_secondary_ec_numbers { + my ($gene_obj) = @_; + return (sort @{$gene_obj->{secondary_ec_numbers}}); +} + + + +=over 4 + +=item add_secondary_gene_names() + +B Adds secondary gene name(s) + +B (gene_name_1, gene_name_2, ....) + +Single gene name or list of gene names is allowed + + +B none + +=back + +=cut + + + +#### +sub add_secondary_gene_names { + my ($gene_obj, @gene_names) = @_; + push (@{$gene_obj->{secondary_gene_names}}, @gene_names); +} + + +=over 4 + +=item add_secondary_product_names() + +B Adds secondary product names + +B (product_name_1, product_name_2, ...) + +Single or list of product names as parameter + +B none + +Primary gene name added directly as an attribute like so + $gene_obj->{gene_name} = name + +=back + +=cut + + +#### +sub add_secondary_product_names { + my ($gene_obj, @product_names) = @_; + &trim_leading_trailing_ws(\@product_names); + push (@{$gene_obj->{secondary_product_names}}, @product_names); +} + + +=over 4 + +=item add_secondary_gene_symbols() + +B Add secondary gene symbols + +B (gene_symbol_1, gene_symbol_2, ...) + +String or list context + +B none + +Primary gene_symbol added directly as attribute like so: + $gene_obj->{gene_sym} = symbol + +=back + +=cut + + +#### +sub add_secondary_gene_symbols { + my ($gene_obj, @gene_symbols) = @_; + &trim_leading_trailing_ws(\@gene_symbols); + push (@{$gene_obj->{secondary_gene_symbols}}, @gene_symbols); +} + + + + + +=over 4 + +=item add_secondary_ec_numbers() + +B Add secondary Enzyme Commission (EC) numbers + +B (EC_1, EC_2, ...) + +String or list context + +B none + + +Primary EC number added directly as an attribute like so: + $gene_obj->{ec_num} = EC_number + +=back + +=cut + + +#### +sub add_secondary_ec_numbers { + my ($gene_obj, @ec_numbers) = @_; + &trim_leading_trailing_ws(\@ec_numbers); + push (@{$gene_obj->{secondary_ec_numbers}}, @ec_numbers); +} + +#### +sub to_alignment_GFF3_format { + my ($gene_obj, $id, $target, $source) = @_; + + unless (defined $source) { + $source = "."; + } + + ## Note, only examines gene_obj and doesn't go deeper into alt-splicing layers, ... send isoforms in as separate objs. + + unless ( (ref $gene_obj) && defined($id) && defined($target)) { + croak "Error, need gene_obj, id, and target names as params"; + } + + my $gff3_alignment_text = ""; + + my $orient = $gene_obj->get_orientation(); + my $scaff = $gene_obj->{asmbl_id}; + + my @exons = sort {$a->{end5}<=>$b->{end5}} $gene_obj->get_exons(); + + if ($orient eq '-') { + @exons = reverse @exons; + } + + + my $match_lend = 0; + + foreach my $exon (@exons) { + + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + + my $m_lend = $match_lend + 1; + my $m_rend = $match_lend + ($rend - $lend + 1); + + + $gff3_alignment_text .= join("\t", $scaff, $source, "match", $lend, $rend, "100", $orient, '.', # giving everything 100% identity since genome-based + "ID=$id;Target=$target $m_lend $m_rend +") . "\n"; + + + $match_lend = $m_rend; + + + } + + return($gff3_alignment_text); +} + + + + +#### +sub to_transcript_GTF_format { + my ($gene_obj) = @_; + + ## no worries about protein-coding regions. Only report transcripts and exons tied to a particular gene. + ## used with cufflinks package for computing FPKM values + + my $gtf_text = ""; + + foreach my $gene ($gene_obj, $gene_obj->get_additional_isoforms()) { + + my $gene_id = $gene->{TU_feat_name} || ""; + my $transcript_id = $gene->{Model_feat_name} || ""; + my $asmbl_id = $gene_obj->{asmbl_id}; + my ($lend, $rend) = sort {$a<=>$b} $gene_obj->get_transcript_span(); + my $orientation = $gene_obj->get_orientation(); + + my $com_name = $gene_obj->{com_name} || ""; + $com_name =~ s/;/_/g; + $com_name =~ s/\"//g; + + + if ($gene->{gene_type} eq "protein-coding") { + my @exons = $gene->get_exons(); + + $gtf_text .= join("\t", $asmbl_id, ".", "transcript", $lend, $rend, ".", $orientation, ".", + "gene_id \"$gene_id\"; transcript_id \"$transcript_id\"; name \"$com_name\";") . "\n"; + + foreach my $exon (@exons) { + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + + $gtf_text .= join("\t", $asmbl_id, ".", "exon", $lend, $rend, ".", $orientation, ".", + "gene_id \"$gene_id\"; transcript_id \"$transcript_id\";") . "\n"; + + } + + } + else { + + ## non-protein-coding features + $gtf_text .= join("\t", $asmbl_id, ".", $gene->{gene_type}, $lend, $rend, ".", $orientation, ".", + "gene_id \"$gene_id\"; transcript_id \"$transcript_id\"; name \"$com_name\";") . "\n"; + + + } + + $gtf_text .= "\n"; + } + + + return($gtf_text); +} + + + +=over 4 + +=item to_GTF_format() + +B Outputs text corresponding to the representation of the gene in GTF format. + +B $genome_seq_ref, %preferences + +B string + + +GTF format is described in "Current Protocols in Bioinformatics(2003)" 4.8.1-4.8.19 +in "Using TWINSCAN to Predict Gene Structures in Genomic DNA Sequences". + +Each line of the GTF format includes the following tab-delimited fields: + +[seqname] [source] [feature] [start] [end] [score] [strand] [frame] [attributes] + +This is further elaborated below: + +[feature] contains one of the following: start_codon, stop_codon, CDS +[attributes] contains 'gene_id' and 'transcript_id' fields. All features of the same transcript should share the same transcript_id value. By default, the TU_feat_name and model_feat_name are used as the gene_id and transcript_id, respectively. + + + Using the %preferences input parameter, the preferred values or gene attributes can be used for seqname, source, gene_id, or transcript_id, each used as a key to the %preferences hash. Given the value of %preferences is a gene attribute, that attribute value will be used, otherwise, the raw value will be used. + +For example: %preferences = ( seqname => 'mySeqname', + gene_id => 'pub_locus' ); + +Would result in 'mySeqname' used in the [seqname] field, and the $gene_obj->{pub_locus} value + +Here are the defaults: +[seqname] = asmbl_id +[source] = annotation +gene_id (TU_feat_name) +transcript_id (Model_feat_name) + +** Partial Genes are NOT Supported ** ( undef is returned ) +** Genes with split start or stop codons are unsupported ** (undef is returned) + +=back + +=cut + + ; + +sub to_GTF_format { + my $gene_obj = shift; + my ($genome_seq_ref, %preferences) = @_; + + unless (ref $genome_seq_ref) { + confess "Error, need genome seq reference as param"; + } + + my $is_pseudogene = $gene_obj->is_pseudogene(); + + + my $TU_feat_name = $gene_obj->{TU_feat_name}; + my $model_feat_name = $gene_obj->{Model_feat_name}; + + # rid whitespace in identifiers + $TU_feat_name =~ s/\s+/_/g; + $model_feat_name =~ s/\s+/_/g; + + my $seqname = $preferences{seqname} || $gene_obj->{asmbl_id}; + my $source = $preferences{source} || $gene_obj->{source} || "."; + + my $gene_id; + if (my $token = $preferences{gene_id}) { + $gene_id = $gene_obj->{$token}; + } else { + $gene_id = $TU_feat_name; + } + + my $transcript_id; + if (my $token = $preferences{model_id}) { + $transcript_id = $gene_obj->{$token}; + } else { + $transcript_id = $model_feat_name; + } + + my @exons = $gene_obj->get_exons(); + my $orientation = $gene_obj->get_orientation(); + my @gtf_text; + + my $gene_obj_for_gtf = $gene_obj; #if got stop codon, will need to strip it off. + my $com_name = $gene_obj->{com_name}; + $com_name =~ s/\s+$// if $com_name; + $com_name =~ s/[\"\']//g if $com_name; + + my $name_txt = ($com_name) ? "Name \"$com_name\";" : ""; + + + ## Gene record + unless ($preferences{'gene_record_already_done'}) { + + my ($gene_lend, $gene_rend) = sort {$a<=>$b} $gene_obj->get_gene_span(); + + push (@gtf_text, [$seqname, + $source, + "gene", + $gene_lend, + $gene_rend, + "0", + $orientation, + ".", + "gene_id \"$gene_id\"; $name_txt"]); + } + + ## Transcript record + my ($trans_lend, $trans_rend) = sort {$a<=>$b} $gene_obj->get_transcript_span(); + push (@gtf_text, [$seqname, + $source, + "transcript", + $trans_lend, + $trans_rend, + "0", + $orientation, + ".", + "gene_id \"$gene_id\"; transcript_id \"$transcript_id\"; $name_txt"]); + + unless ($is_pseudogene) { + $gene_obj->set_CDS_phases($genome_seq_ref); + + + ## check for start and stop codons. + my $cds_seq = uc $gene_obj->create_CDS_sequence($genome_seq_ref); + my @stop_codons = &Nuc_translator::get_stop_codons(); + + my $first_CDS_segment = $gene_obj->get_first_CDS_segment(); + my $first_phase = $first_CDS_segment->get_phase(); + my $cds_is_integral_codon_num = (length($cds_seq) % 3 == 0) ? 1 : 0; + + ## examine start codon: + my $init_codon = substr($cds_seq, 0, 3); + if ($first_phase == 0 && $init_codon eq 'ATG') { # got start codon. + my @start_coordsets = $gene_obj->get_start_codon_coordinates(); + foreach my $start_pair (@start_coordsets) { + my ($start_lend, $start_rend) = sort {$a<=>$b} @$start_pair; + push (@gtf_text, [$seqname, + $source, + "start_codon", + $start_lend, + $start_rend, + "0", + $orientation, + "0", + "gene_id \"$gene_id\"; transcript_id \"$transcript_id\"; $name_txt"]); + } + } + + my $candidate_stop_codon = uc substr($cds_seq, length($cds_seq) - 3, 3); + my @found_stop = grep { $_ eq $candidate_stop_codon } @stop_codons; + + if (@found_stop) { + # got a stop codon. + # check to see that the stop codon is in-frame. + if ((length($cds_seq) - $first_phase) % 3 == 0) { # yes, stop is in frame. + + my @stop_codon_coords = $gene_obj->get_stop_codon_coords(); + foreach my $stop_pair (@stop_codon_coords) { + my ($stop_lend, $stop_rend) = sort {$a<=>$b} @$stop_pair; + + push (@gtf_text, [$seqname, + $source, + "stop_codon", + $stop_lend, + $stop_rend, + "0", + $orientation, + "0", + "gene_id \"$gene_id\"; transcript_id \"$transcript_id\"; $name_txt"]); + } + + $gene_obj_for_gtf = $gene_obj->clone_gene(); + + $gene_obj_for_gtf->trim_stop_codon(); + + } + } + + } + + ## report the exons and CDS regions: + foreach my $exon ($gene_obj_for_gtf->get_exons()) { + + my ($exon_lend, $exon_rend) = sort {$a<=>$b} $exon->get_coords(); + + push (@gtf_text, [$seqname, + $source, + "exon", + $exon_lend, + $exon_rend, + "0", + $orientation, + ".", + "gene_id \"$gene_id\"; transcript_id \"$transcript_id\"; $name_txt"]); + + + + my $cds = ($is_pseudogene) ? $exon : $exon->get_CDS_exon_obj(); + + if ($cds) { + my $phase = "."; + unless ($is_pseudogene) { + $phase = $cds->get_phase(); + if ($phase) { + $phase = ($phase == 1) ? 2 : 1; # reverse it according to GFF3 vs. GTF representation. + } + } + + my ($cds_lend, $cds_rend) = sort {$a<=>$b} $cds->get_coords(); + + push (@gtf_text, [$seqname, + $source, + "CDS", + $cds_lend, + $cds_rend, + "0", + $orientation, + "$phase", + "gene_id \"$gene_id\"; transcript_id \"$transcript_id\"; $name_txt"]); + } + + + } + + unless ($is_pseudogene) { + + ## Get UTR info: + { + for my $pair ($gene_obj->get_3prime_UTR_coords) { + my ($lend,$rend) = sort {$a<=>$b} @$pair; + push (@gtf_text, [$seqname, + $source, + "3UTR", + $lend, + $rend, + "0", + $orientation, + "0", + "gene_id \"$gene_id\"; transcript_id \"$transcript_id\"; $name_txt"] ); + } + for my $pair ($gene_obj->get_5prime_UTR_coords) { + my ($lend,$rend) = sort {$a<=>$b} @$pair; + push (@gtf_text, [$seqname, + $source, + "5UTR", + $lend, + $rend, + "0", + $orientation, + "0", + "gene_id \"$gene_id\"; transcript_id \"$transcript_id\"; $name_txt" ] ); + } + } + + } + + @gtf_text = sort {$a->[3] <=> $b->[3]} @gtf_text; + + if ($orientation eq '-') { + @gtf_text = reverse @gtf_text; + } + + my $GTF = ""; + foreach my $gtf_row (@gtf_text) { + $GTF .= join ("\t", @$gtf_row) . "\n"; + } + + foreach my $isoform ($gene_obj->get_additional_isoforms()) { + my %iso_pref = %preferences; + $iso_pref{'gene_record_already_done'} = 1; + $GTF .= "\n" . $isoform->to_GTF_format($genome_seq_ref, %iso_pref); + } + + return ($GTF); +} + + + + + +#### +sub get_start_codon_coordinates { + my $gene_obj = shift; + + my $orient = $gene_obj->get_orientation(); + + ## just want the coordinate pairs that define the first three CDS bases. + + my @cds_coords; + foreach my $exon ($gene_obj->get_exons()) { + if (my $cds = $exon->get_CDS_exon_obj()) { + my ($cds_end5, $cds_end3) = $cds->get_coords(); + push (@cds_coords, [$cds_end5, $cds_end3]); + } + } + + my @start_coords; + my $start_len_want = 3; + foreach my $cds_coordpair (@cds_coords) { + my ($cds_end5, $cds_end3) = @$cds_coordpair; + my $cds_seg_len = abs ($cds_end3 - $cds_end5) + 1; + + my $extract_len = ($cds_seg_len < $start_len_want) ? $cds_seg_len : $start_len_want; + if ($orient eq '+') { + push (@start_coords, [$cds_end5, $cds_end5 + $extract_len - 1]); + } + else { + push (@start_coords, [$cds_end5, $cds_end5 - $extract_len + 1]); + } + $start_len_want -= $extract_len; + + if ($start_len_want <= 0) { last; } + } + + if ($start_len_want > 0) { + confess "Error, trouble extracting start codon coordinates from cds coordsets: " . Dumper (\@cds_coords); + } + + return (@start_coords); +} + + + + +#### +sub get_stop_codon_coords { + my $gene_obj = shift; + + my $orient = $gene_obj->get_orientation(); + + ## just want the coordinate pairs that define the last three CDS bases. + + my @cds_coords; + foreach my $exon (reverse $gene_obj->get_exons()) { + if (my $cds = $exon->get_CDS_exon_obj()) { + my ($cds_end5, $cds_end3) = $cds->get_coords(); + push (@cds_coords, [$cds_end5, $cds_end3]); + } + } + + my @stop_coords; + my $stop_len_want = 3; + foreach my $cds_coordpair (@cds_coords) { + my ($cds_end5, $cds_end3) = @$cds_coordpair; + my $cds_seg_len = abs ($cds_end3 - $cds_end5) + 1; + + my $extract_len = ($cds_seg_len < $stop_len_want) ? $cds_seg_len : $stop_len_want; + if ($orient eq '+') { + push (@stop_coords, [$cds_end3 - $extract_len + 1, $cds_end3]); + } + else { + push (@stop_coords, [$cds_end3, $cds_end3 + $extract_len - 1]); + } + $stop_len_want -= $extract_len; + + if ($stop_len_want <= 0) { last; } + } + + if ($stop_len_want > 0) { + confess "Error, trouble extracting stop codon coordinates from cds coordsets: " . Dumper (\@cds_coords); + } + + + return (@stop_coords); + +} + + +#### +sub trim_stop_codon { + my $gene_obj = shift; + + ## just trimming the last three bases from the CDS's, changing the current gene object. + + my @exons = reverse $gene_obj->get_exons(); + + my $orient = $gene_obj->get_orientation(); + + my $stop_len_want = 3; + foreach my $exon (@exons) { + if (my $cds = $exon->get_CDS_exon_obj()) { + + my ($cds_end5, $cds_end3) = $cds->get_coords(); + my $cds_seg_len = abs ($cds_end3 - $cds_end5) + 1; + + my $extract_len = ($cds_seg_len < $stop_len_want) ? $cds_seg_len : $stop_len_want; + + if ($cds_seg_len == $extract_len) { + # delete it! + $exon->delete_CDS_exon_obj(); + } + else { + ## truncate it by extract_len + if ($orient eq '+') { + $cds->{end3} -= $extract_len; + } + + else { + $cds->{end3} += $extract_len; + } + } + $stop_len_want -= $extract_len; + + if ($stop_len_want <= 0) { last; } + } + } + if ($stop_len_want > 0) { + confess "Error, trouble extracting all stop codon coordinates from cds coordsets. " . $gene_obj->toString(); + } + + return; + +} + + + +=over 4 + +=item to_GFF3_format() + +B Outputs text corresponding to the representation of the gene in GFF3 format (still under development). + +B + +B string + +GFF3 defined at: +http://song.sourceforge.net/gff3-jan04.shtml + +(some text lifted from above site provided below for reference purposes) + +The format consists of 9 columns, separated by tabs or spaces. The +following unescaped characters are allowed within fields: +[a-zA-Z0-9.:^*$@!+_?-]. All other characters must must be escaped +using the URL escaping conventions. Unescaped quotation marks, +backslashes and other ad-hoc escaping conventions that have been added +to the GFF format are explicitly forbidden. The =, ; and % characters +have reserved meanings as described below, and must be escaped when +used in other contexts. + +Undefined fields are replaced with the "." character, as described in +the original GFF spec. + +Column 1: "seqid" + +The ID of the landmark used to establish the coordinate system for the +current feature. IDs must contain alphanumeric characters. +Whitespace, if present, must be escaped using URL escaping rule +(e.g. space="%20" or "+"). Sequences must *NOT* begin with an +unescaped ">". + +Column 2: "source" + +The source of the feature. This is unchanged from the older GFF specs +and is not part of a controlled vocabulary. + +Column 3: "type" + +The type of the feature (previously called the "method"). This is +constrained to be either: (a) a term from the "lite" sequence +ontology, SOFA; or (b) a SOFA accession number. The latter +alternative is distinguished using the syntax SO:000000. + +Columns 4 & 5: "start" and "end" + +The start and end of the feature, in 1-based integer coordinates, +relative to the landmark given in column 1. Start is always less than +or equal to end. + +For zero-length features, such as insertion sites, start equals end +and the implied site is to the right of the indicated base. This +convention holds regardless of the strandedness of the feature. + +Column 6: "score" + +The score of the feature, a floating point number. As in earlier +versions of the format, the semantics of the score are ill-defined. +It is strongly recommended that E-values be used for sequence +similarity features, and that P-values be used for ab initio gene +prediction features. + +Column 7: "strand" + +The strand of the feature. + for positive strand (relative to the +landmark), - for minus strand, and . for features that are not +stranded. In addition, ? can be used for features whose strandedness +is relevant, but unknown. + +Column 8: "phase" + +For features of type "exon", the phase indicates where the feature +begins with reference to the reading frame. The phase is one of the +integers 0, 1,or 2, indicating that the first base of the feature +corresponds to the first, second or last base of the codon, +respectively. This is NOT to be confused with the frame, but relates +to the relative position of the translational start in whatever strand +the feature is in. + +Column 9: "attributes" + +A list of feature attributes in the format tag=value. Multiple +tag=value pairs are separated by semicolons. URL escaping rules are +used for tags or values containing the following characters: ",=;". +Whitespace should be replaced with the "+" character or the %20 URL +escape. This will allow the file to survive text processing programs +that convert tabs into spaces. + +These tags have predefined meanings: + + ID Indicates the name of the feature. IDs must be unique + within the scope of the GFF file. + + Name Display name for the feature. This is the name to be + displayed to the user. Unlike IDs, there is no requirement + that the Name be unique within the file. + + Alias A secondary name for the feature. It is suggested that + this tag be used whenever a secondary identifier for the + feature is needed, such as locus names and + accession numbers. Unlike ID, there is no requirement + that Alias be unique within the file. + + Parent Indicates the parent of the feature. A parent ID can be + used to group exons into transcripts, transcripts into + genes, an so forth. A feature may have multiple parents. + + Target Indicates the target of a nucleotide-to-nucleotide or + protein-to-nucleotide alignment. The format of the + value is "target_id+start+end". + + Gap The alignment of the feature to the target if the two are + not colinear (e.g. contain gaps). The alignment format is + taken from the CIGAR format described in the + Exonerate documentation. + (http://cvsweb.sanger.ac.uk/cgi-bin/cvsweb.cgi/exonerate + ?cvsroot=Ensembl). See "THE GAP ATTRIBUTE" for a description + of this format. + + Note A free text note. + + Dbxref A database cross reference. See the section + "Ontology Associations and Db Cross References" for + details on the format. + + Ontology_term A cross reference to an ontology term. See + the section "Ontology Associations and Db Cross References" + for details. + +Multiple attributes of the same type are indicated by separating the +values with the comma "," character, as in: + + Parent=AF2312,AB2812,abc-3 + +Note that attribute names are case sensitive. "Parent" is not the +same as "parent". + +All attributes that begin with an uppercase letter are reserved for +later use. Attributes that begin with a lowercase letter can be used +freely by applications. + + + +=back + +=cut + + ; + + + +sub to_GFF3_format { + my ($gene_obj, %preferences) = @_; + + my $gene_id = $gene_obj->{TU_feat_name}; + if ($gene_id =~ /;/) { + $gene_id = "\"$gene_id\""; + } + + my $strand = $gene_obj->get_orientation(); + + my @noteText; + + if ($gene_obj->{is_pseudogene}) { + push (@noteText, "(pseudogene)"); + } + + ## parse preferences + my $asmbl_id = $preferences{seqid} || $gene_obj->{asmbl_id}; + my $source = $preferences{source} || $gene_obj->{source} || "."; + + unless (defined $asmbl_id) { + confess "Error, no asmbl_id from gene_obj\n"; + } + + + my ($gene_lend, $gene_rend) = sort {$a<=>$b} $gene_obj->get_gene_span(); + my $com_name = $gene_obj->{com_name}; + unless ($com_name =~ /\w/) { + $com_name = ""; + } + + if ($com_name) { + if ($preferences{uri_encode_name}) { + # uri escape it: + use URI::Escape; + $com_name = uri_escape($com_name); + } + else { + unless (substr($com_name,0,1) =~ /\'|\"/ && substr($com_name, -1, 1) =~ /\'|\"/) { + $com_name = "\"$com_name\""; + } + } + } + + my $gene_alias = ""; + if (my $pub_locus = $gene_obj->{pub_locus}) { + $gene_alias = "Alias=$pub_locus;"; + } + + my $feat_type = ($gene_obj->{gene_type} eq "protein-coding") ? "gene" : $gene_obj->{gene_type}; + + + my $gff3_text = "$asmbl_id\t$source\t$feat_type\t$gene_lend\t$gene_rend\t.\t$strand\t.\tID=$gene_id;Name=$com_name;$gene_alias\n"; ## note, non-coding gene features are currently represented by a simple single coordinate pair. + + if ($gene_obj->{gene_type} eq "protein-coding") { + + my $gene_obj_ref = $gene_obj; + + foreach my $gene_obj ($gene_obj_ref, $gene_obj_ref->get_additional_isoforms() ) { + + my $model_id = $gene_obj->{Model_feat_name}; + if ($model_id =~ /;/) { + $model_id = "\"$model_id\""; + } + + my $model_alias = ""; + if (my $model_locus = $gene_obj->{Model_pub_locus}) { + $model_alias = "Alias=$model_locus;"; + } + + my ($mrna_lend, $mrna_rend) = $gene_obj->get_transcript_span(); + + $gff3_text .= "$asmbl_id\t$source\tmRNA\t$mrna_lend\t$mrna_rend\t.\t$strand\t.\tID=$model_id;Parent=$gene_id;Name=$com_name;$model_alias\n"; + + ## mark the first and last CDS entries (for now, an unpleasant hack!) + my @exons = $gene_obj->get_exons(); + ## find the first cds + foreach my $exon (@exons) { + if (my $cds = $exon->get_CDS_obj()) { + $cds->{first_cds} = 1; + last; + } + } + @exons = reverse @exons; + foreach my $exon (@exons) { + if (my $cds = $exon->get_CDS_obj()) { + $cds->{last_cds} = 1; + last; + } + } + + my $prime5_partial = $gene_obj->is_5prime_partial(); + my $prime3_partial = $gene_obj->is_3prime_partial(); + + + ## annotate 5' utr + if ($gene_obj->has_CDS() && $gene_obj->has_5prime_UTR()) { + my @prime5_utr = $gene_obj->get_5prime_UTR_coords(); + if (@prime5_utr) { + my $utr_count = 0; + foreach my $coordset (@prime5_utr) { + my ($lend, $rend) = sort {$a<=>$b} @$coordset; + $utr_count++; + my $utr_id = "$model_id.utr5p$utr_count"; + $gff3_text .= "$asmbl_id\t$source\tfive_prime_UTR\t$lend\t$rend\t.\t$strand\t.\tID=$utr_id;Parent=$model_id\n"; + } + } + } + + + my $exon_counter = 0; + foreach my $exon ($gene_obj->get_exons()) { + $exon_counter++; + my ($exon_lend, $exon_rend) = sort {$a<=>$b} $exon->get_coords(); + my $exon_ID_string = ""; + if (my $exon_feat_name = $exon->{feat_name}) { + $exon_ID_string = "$exon_feat_name"; + } + else { + $exon_ID_string = "$model_id.exon$exon_counter"; + } + $gff3_text .= "$asmbl_id\t$source\texon\t$exon_lend\t$exon_rend\t.\t$strand\t.\tID=${exon_ID_string};Parent=$model_id\n"; + + if (my $cds_obj = $exon->get_CDS_obj()) { + my ($cds_lend, $cds_rend) = sort {$a<=>$b} $cds_obj->get_coords(); + my $phase = $cds_obj->{phase}; + if (defined($phase)) { + ## use GFF3 definition of phase, which is how many bases to trim before encountering first base of start + if ($phase == 2) { + $phase = 1; + } + elsif ($phase == 1) { + $phase = 2; + } + # phase 0 remains 0 + } + else { + $phase = "."; #use phase info if avail + } + + + my $cds_ID_string = "cds.$model_id"; + + # according to the GFF3 spec, CDS segments from the same coding region should have the same identifier. + #if (my $cds_feat_name = $cds_obj->{feat_name}) { + # $cds_ID_string = "$cds_feat_name"; + #} + #else { + # $cds_ID_string = "$model_id.cds$exon_counter"; + #} + + my $partial_text = ""; + if ($prime5_partial && $cds_obj->{first_cds}) { + $partial_text .= ";5_prime_partial=true"; + } + if ($prime3_partial && $cds_obj->{last_cds}) { + $partial_text .= ";3_prime_partial=true"; + } + + $gff3_text .= "$asmbl_id\t$source\tCDS\t$cds_lend\t$cds_rend\t.\t$strand\t$phase\tID=${cds_ID_string};Parent=$model_id$partial_text\n"; + } + } + + ## annotate 3' utr + if ($gene_obj->has_CDS() && $gene_obj->has_3prime_UTR()) { + my @prime3_utr = $gene_obj->get_3prime_UTR_coords(); + if (@prime3_utr) { + my $utr_count = 0; + foreach my $coordset (@prime3_utr) { + my ($lend, $rend) = sort {$a<=>$b} @$coordset; + $utr_count++; + my $utr_id = "$model_id.utr3p$utr_count"; + $gff3_text .= "$asmbl_id\t$source\tthree_prime_UTR\t$lend\t$rend\t.\t$strand\t.\tID=$utr_id;Parent=$model_id\n"; + } + } + + } + } + + } ## end of protein-coding genes + + + ## strip off any trailing whitespace and semicolons: + my @lines = split (/\n/, $gff3_text); + foreach my $line (@lines) { + $line =~ s/\s+$//; + $line =~ s/;$//; + } + + $gff3_text = join ("\n", @lines) . "\n"; + + return ($gff3_text); + +} + + + +=over 4 + +=item to_BED_format() + +B describes gene in BED format +B (uri_encode => 1|0) +B string + + +BED format described here: +http://genome.ucsc.edu/FAQ/FAQformat.html#format1 + + BED format + + + + +BED format provides a flexible way to define the data lines that are displayed in an annotation track. BED lines have three required fields and nine additional optional fields. The number of fields per line must be consistent throughout any single set of data in an annotation track. The order of the optional fields is binding: lower-numbered fields must always be populated if higher-numbered fields are used. + +The first three required BED fields are: + +1. chrom - The name of the chromosome (e.g. chr3, chrY, chr2_random) or scaffold (e.g. scaffold10671). + +2. chromStart - The starting position of the feature in the chromosome or scaffold. The first base in a chromosome is numbered 0. + +3. chromEnd - The ending position of the feature in the chromosome or scaffold. The chromEnd base is not included in the display of the feature. For example, the first 100 bases of a chromosome are defined as chromStart=0, chromEnd=100, and span the bases numbered 0-99. + +The 9 additional optional BED fields are: + +4. name - Defines the name of the BED line. This label is displayed to the left of the BED line in the Genome Browser window when the track is open to full display mode or directly to the left of the item in pack mode. + +5. score - A score between 0 and 1000. If the track line useScore attribute is set to 1 for this annotation data set, the score value will determine the level of gray in which this feature is displayed (higher numbers = darker gray). This table shows the Genome Browsers translation of BED score values into shades of gray + +6. strand - Defines the strand - either '+' or '-'. + +7. thickStart - The starting position at which the feature is drawn thickly (for example, the start codon in gene displays). + +8. thickEnd - The ending position at which the feature is drawn thickly (for example, the stop codon in gene displays). + +9. itemRgb - An RGB value of the form R,G,B (e.g. 255,0,0). If the track line itemRgb attribute is set to "On", this RBG value will determine the display color of the data contained in this BED line. NOTE: It is recommended that a simple color scheme (eight colors or less) be used with this attribute to avoid overwhelming the color resources of the Genome Browser and your Internet browser. + +10. blockCount - The number of blocks (exons) in the BED line. + +11. blockSizes - A comma-separated list of the block sizes. The number of items in this list should correspond to blockCount. + +12. blockStarts - A comma-separated list of block starts. All of the blockStart positions should be calculated relative to chromStart. The number of items in this list should correspond to blockCount. + +Example: +Heres an example of an annotation track that uses a complete BED definition: +track name=pairedReads description="Clone Paired Reads" useScore=1 +chr22 1000 5000 cloneA 960 + 1000 5000 0 2 567,488, 0,3512 +chr22 2000 6000 cloneB 900 - 2000 6000 0 2 433,399, 0,3601 + + + + +=cut + + +sub to_BED_format { + my $self = shift; + my %params = @_; + + my $strand = $self->get_strand(); + + my ($coding_lend, $coding_rend) = sort {$a<=>$b} $self->get_CDS_span(); + + my $scaffold = $self->{asmbl_id}; + + my $gene_id = $self->{TU_feat_name}; + my $trans_id = $self->{Model_feat_name}; + + my $com_name = $self->{com_name} || ""; + + my $score = $params{score} || 0; + + if (my $alias = $self->{pub_locus}) { + $com_name = "Alias=$alias;$com_name"; + } + + + if ($gene_id) { + $com_name = "$gene_id;$com_name"; + } + + if ($trans_id) { + $com_name = "ID=$trans_id;$com_name"; + } + else { + $com_name = "ID=$com_name"; + } + + if ($params{uri_encode}) { + $com_name = uri_escape($com_name); + } + + + my @exons = sort {$a->{end5}<=>$b->{end5}} $self->get_exons(); + + my @exon_coords; + foreach my $exon (@exons) { + + my ($exon_lend, $exon_rend) = sort {$a<=>$b} $exon->get_coords(); + push (@exon_coords, [$exon_lend, $exon_rend]); + } + + + my @starts; + my @lengths; + + my $gene_lend = $exon_coords[0]->[0]; + my $gene_rend = $exon_coords[$#exon_coords]->[1]; + + foreach my $exon_coordset (@exon_coords) { + my ($exon_lend, $exon_rend) = @$exon_coordset; + + my $start = $exon_lend - $gene_lend; + push (@starts, $start); + + my $length = $exon_rend - $exon_lend + 1; + push (@lengths, $length); + } + + + ## construct bed output. + + $com_name =~ s/ /_/g; + + my $bed_line = join("\t", $scaffold, + $gene_lend-1, $gene_rend, + $com_name, + $score, + $strand, + $coding_lend-1, $coding_rend, + "0", # rgb info - use '.' to allow user customization in IGV. Need 0 for compatibility with UCSC browser. + scalar(@lengths), + join(",", @lengths), + join(",", @starts) + ) . "\n"; + + foreach my $isoform ($self->get_additional_isoforms()) { + $bed_line .= $isoform->to_BED_format(%params); + } + + return($bed_line); +} + + + +# static method, returns gene object. +sub BED_line_to_gene_obj { + my ($bed_line) = @_; + + if (ref $bed_line) { + confess "Error, static method, just provide bed text line, returns gene_obj"; + } + + + my @x = split(/\t/, $bed_line); + + my $scaff = $x[0]; + my $gene_lend = $x[1] + 1; + my $gene_rend = $x[2]; + + my $com_name = $x[3]; + + my $score = $x[4]; + my $orient = $x[5]; + + if ($orient eq '*') { + $orient = '+'; + } + + + my $coding_lend = $x[6] + 1; + my $coding_rend = $x[7]; + + my $rgb_color = $x[8]; + + my $num_exons = $x[9]; + + my $lengths_text = $x[10]; + my $exon_relative_starts_text = $x[11]; + + my @lengths = split(/,/, $lengths_text); + my @exon_relative_starts = split(/,/, $exon_relative_starts_text); + + my @exons; + + while (@lengths) { + my $len = shift @lengths; + my $start = shift @exon_relative_starts; + + my $exon_lend = $gene_lend + $start; + my $exon_rend = $exon_lend + $len - 1; + + + print "Len: $len, start=$start ====> $exon_lend - $exon_rend\n" if $DEBUG; + + push (@exons, [$exon_lend, $exon_rend]); + + } + + + print "Coding: $coding_lend-$coding_rend, Exons: " . Dumper (\@exons) if $DEBUG; + + my $gene_obj = new Gene_obj(); + $gene_obj->build_gene_obj_exons_n_cds_range(\@exons, $coding_lend, $coding_rend, $orient); + + $gene_obj->{com_name} = $com_name; + $gene_obj->{asmbl_id} = $scaff; + + $com_name =~ s/\s+/\|/g; # reformat as an identifier with no whitespace + + $gene_obj->{TU_feat_name} = "$com_name"; + $gene_obj->{Model_feat_name} = "m.$com_name"; + + return($gene_obj); + + +} + + + + + + +## Private, remove leading and trailing whitespace characters: +sub trim_leading_trailing_ws { + my ($ref) = @_; + if (ref $ref eq "SCALAR") { + $$ref =~ s/^\s+|\s+$//g; + } elsif (ref $ref eq "ARRAY") { + foreach my $element (@$ref) { + $element =~ s/^\s+|\s+$//g; + } + } else { + my $type = ref $ref; + die "Currently don't support trim_leading_trailing_ws(ref type: $type)\n"; + } +} + + + +=over 4 + +=item to_GTF2_format() + +B provides gene in GTF2 format + +B genome_seq_ref, [properties_href] + +B text + + +properties_href encodes preferences like so + + properties_href = { + seqname => tigr_asmbl_id_1000, # by default, asmbl_id is used as encoded in gene_obj + + source => MyGenePrediction, # by default, set to "TIGR" + + include_comments => 0, # turned on by default, indicating partial or pseudogenes with preceding comment lines + + } + + + +The GTF2 format is described here: +http://genes.cs.wustl.edu/GTF2.html + +as follows: + +GTF2 format (Revised Ensembl GTF) +Gene transfer format. This borrows from GFF, but has additional structure that warrants a separate definition and format name. +NEW! Validating Parser for GTF + +Structure is as GFF, so the fields are: + [attributes] [comments] + +Here is a simple example with 3 translated exons. Order of rows is not important. + +AB000381 Twinscan CDS 380 401 . + 0 gene_id "001"; transcript_id "001.1"; +AB000381 Twinscan CDS 501 650 . + 2 gene_id "001"; transcript_id "001.1"; +AB000381 Twinscan CDS 700 707 . + 2 gene_id "001"; transcript_id "001.1"; +AB000381 Twinscan start_codon 380 382 . + 0 gene_id "001"; transcript_id "001.1"; +AB000381 Twinscan stop_codon 708 710 . + 0 gene_id "001"; transcript_id "001.1"; + +The whitespace in this example is provided only for readability. In GTF, fields must be separated by a single TAB and no white space. + + +The FPC contig ID from the Golden Path. + + +The source column should be a unique label indicating where the annotations came from --- typically the name of either a prediction program or a public database. + + +The following feature types are required: "CDS", "start_codon", "stop_codon". The feature "exon" is optional, since this project will not evaluate predicted splice sites outside of protein coding regions. All other features will be ignored. + +CDS represents the coding sequence starting with the first translated codon and proceeding to the last translated codon. Unlike Genbank annotation, the stop codon is not included in the CDS for the terminal exon. + + +Integer start and end coordinates of the feature relative to the beginning of the sequence named in . must be less than or equal to . Sequence numbering starts at 1. Values of and that extend outside the reference sequence are technically acceptable, but they are discouraged for purposes of this project. + + +The score field will not be used for this project, so you can either provide a meaningful float or replace it by a dot. + + +0 indicates that the first whole codon of the reading frame is located at 5'-most base. 1 means that there is one extra base before the first codon and 2 means that there are two extra bases before the first codon. Note that the frame is not the length of the CDS mod 3. + +Here are the details excised from the GFF spec. Important: Note comment on reverse strand. + + '0' indicates that the specified region is in frame, i.e. that its first base corresponds to the first base of a codon. '1' indicates that there is one extra base, i.e. that the second base of the region corresponds to the first base of a codon, and '2' means that the third base of the region is the first base of a codon. If the strand is '-', then the first base of the region is value of , because the corresponding coding region will run from to on the reverse strand. + +[attributes] +All four features have the same two mandatory attributes at the end of the record: + + * gene_id value; A globally unique identifier for the genomic source of the transcript + * transcript_id value; A globally unique identifier for the predicted transcript. + +These attributes are designed for handling multiple transcripts from the same genomic region. Any other attributes or comments must appear after these two and will be ignored. + +Attributes must end in a semicolon which must then be separated from the start of any subsequent attribute by exactly one space character (NOT a tab character). + +Textual attributes should be surrounded by doublequotes. + +Here is an example of a gene on the negative strand. Larger coordinates are 5' of smaller coordinates. Thus, the start codon is 3 bp with largest coordinates among all those bp that fall within the CDS regions. Similarly, the stop codon is the 3 bp with coordinates just less than the smallest coordinates within the CDS regions. + +AB000123 Twinscan CDS 193817 194022 . - 2 gene_id "AB000123.1"; transcript_id "AB00123.1.2"; +AB000123 Twinscan CDS 199645 199752 . - 2 gene_id "AB000123.1"; transcript_id "AB00123.1.2"; +AB000123 Twinscan CDS 200369 200508 . - 1 gene_id "AB000123.1"; transcript_id "AB00123.1.2"; +AB000123 Twinscan CDS 215991 216028 . - 0 gene_id "AB000123.1"; transcript_id "AB00123.1.2"; +AB000123 Twinscan start_codon 216026 216028 . - . gene_id "AB000123.1"; transcript_id "AB00123.1.2"; +AB000123 Twinscan stop_codon 193814 193816 . - . gene_id "AB000123.1"; transcript_id "AB00123.1.2"; + +Note the frames of the coding exons. For example: + + 1. The first CDS (from 216028 to 215991) always has frame zero. + 2. Frame of the 1st CDS =0, length =38. (frame - length) % 3 = 1, the frame of the 2nd CDS. + 3. Frame of the 2nd CDS=1, length=140. (frame - length) % 3 = 2, the frame of the 3rd CDS. + 4. Frame of the 3rd CDS=2, length=108. (frame - length) % 3 = 2, the frame of the terminal CDS. + 5. Alternatively, the frame of terminal CDS can be calculated without the rest of the gene. Length of the terminal CDS=206. length % 3 =2, the frame of the terminal CDS. + +Here is an example in which the "exon" feature is used. It is a 5 exon gene with 3 translated exons. + +AB000381 Twinscan exon 150 200 . + . gene_id "AB000381.000"; transcript_id "AB000381.000.1"; +AB000381 Twinscan exon 300 401 . + . gene_id "AB000381.000"; transcript_id "AB000381.000.1"; +AB000381 Twinscan CDS 380 401 . + 0 gene_id "AB000381.000"; transcript_id "AB000381.000.1"; +AB000381 Twinscan exon 501 650 . + . gene_id "AB000381.000"; transcript_id "AB000381.000.1"; +AB000381 Twinscan CDS 501 650 . + 2 gene_id "AB000381.000"; transcript_id "AB000381.000.1"; +AB000381 Twinscan exon 700 800 . + . gene_id "AB000381.000"; transcript_id "AB000381.000.1"; +AB000381 Twinscan CDS 700 707 . + 2 gene_id "AB000381.000"; transcript_id "AB000381.000.1"; +AB000381 Twinscan exon 900 1000 . + . gene_id "AB000381.000"; transcript_id "AB000381.000.1"; +AB000381 Twinscan start_codon 380 382 . + 0 gene_id "AB000381.000"; transcript_id "AB000381.000.1"; +AB000381 Twinscan stop_codon 708 710 . + 0 gene_id "AB000381.000"; transcript_id "AB000381.000.1"; + + + + + +=back + +=cut + + + +sub to_GTF2_format () { + my $self = shift; + my $genomic_seq_ref = shift; + + my $properties_href = shift; + unless ($properties_href) { + $properties_href = {}; + } + + + ## need to adjust my frame definition so it's consistent with requirements above in spec. + my $frame_convert = sub { + my $phase = shift; + + my %frame = ( 0 => 0, + 1 => 2, + 2 => 1 ); + return ($frame{$phase}); + }; + + + my $gtf2_text = ""; + + my $gene_obj = $self; + + my $asmbl_id = $properties_href->{seqname} || $gene_obj->{asmbl_id} || die "Error, no asmbl_id as gene_obj att"; + + my $source = $properties_href->{source} || "TIGR"; + + my $gene_id = $gene_obj->{TU_feat_name}; + my $model_id = $gene_obj->{Model_feat_name}; + my $strand = $gene_obj->get_orientation(); + + + my $comment_line = ""; + if ($gene_obj->is_pseudogene()) { + $comment_line .= "$model_id=pseudogene "; + } + + if ( $gene_obj->{gene_type} eq "protein-coding") { + + if (! $gene_obj->is_pseudogene()) { + + $gene_obj->set_CDS_phases($genomic_seq_ref); + # also resets the 5' and 3' partiality attributes based on the longest orf. + + + if ($gene_obj->is_5prime_partial()) { + $comment_line .= "$model_id=5'partial "; + } + else { + $gene_obj->validate_start_codon(); + } + + if ($gene_obj->is_3prime_partial() ) { + $comment_line .= "$model_id=3'partial "; + } + else { + $gene_obj->validate_stop_codon(); + } + } + + my @stop_codon_objs; + my @start_codons; + + if (! $gene_obj->is_pseudogene()) { + + if (! $gene_obj->is_3prime_partial()) { + @stop_codon_objs = $gene_obj->_remove_stop_codons(); + + unless (@stop_codon_objs) { + confess $gene_obj->toString() . "Error, no stop codon objs retrieved for non 3' partial gene"; + } + } + if (! $gene_obj->is_5prime_partial()) { + @start_codons = $self->_extract_start_codons(); + + unless (@start_codons) { + confess $gene_obj->toString() . "Error, no start codon extracted for non 5'partial gene."; + } + } + + } + + foreach my $start_codon (@start_codons) { + my ($start_lend, $start_rend) = sort {$a<=>$b} $start_codon->get_coords(); + my $phase = &$frame_convert($start_codon->{phase}); + $gtf2_text .= "$asmbl_id\t$source\tstart_codon\t$start_lend\t$start_rend\t.\t$strand\t$phase\tgene_id \"$gene_id\"; transcript_id \"$model_id\";\n"; + } + + + foreach my $exon ($gene_obj->get_exons()) { + my ($exon_lend, $exon_rend) = sort {$a<=>$b} $exon->get_coords(); + $gtf2_text .= "$asmbl_id\t$source\texon\t$exon_lend\t$exon_rend\t.\t$strand\t.\tgene_id \"$gene_id\"; transcript_id \"$model_id\";\n"; + + if ($gene_obj->is_pseudogene()) { next; } # don't bother trying to report nonsensical CDSs. + + if (my $cds_obj = $exon->get_CDS_obj()) { + my ($cds_lend, $cds_rend) = sort {$a<=>$b} $cds_obj->get_coords(); + my $phase = $cds_obj->{phase}; + unless (defined($phase)) { + die "Error, no phase defined for cds($cds_lend-$cds_rend) of gene" . $gene_obj->toString(); + } + $phase = &$frame_convert($phase); + + $gtf2_text .= "$asmbl_id\t$source\tCDS\t$cds_lend\t$cds_rend\t.\t$strand\t$phase\tgene_id \"$gene_id\"; transcript_id \"$model_id\";\n"; + } + } + + foreach my $stop_codon (@stop_codon_objs) { + my ($stop_lend, $stop_rend) = sort {$a<=>$b} $stop_codon->get_coords(); + my $phase = &$frame_convert($stop_codon->{phase}); + $gtf2_text .= "$asmbl_id\t$source\tstop_codon\t$stop_lend\t$stop_rend\t.\t$strand\t$phase\tgene_id \"$gene_id\"; transcript_id \"$model_id\";\n"; + } + + foreach my $isoform ($gene_obj->get_additional_isoforms() ) { + $gtf2_text .= $isoform->to_GTF2_format($genomic_seq_ref, $properties_href); + } + } + + if ($comment_line) { + # prefix with \# to actually comment it in the file + $comment_line = "#$comment_line\n"; + } + + my $comment_flag = $properties_href->{include_comments}; + if (defined ($comment_flag) && $comment_flag == 0) { + $comment_line = ""; # clear it + } + + + return ($comment_line . $gtf2_text); +} + + +sub _extract_start_codons { + my $self = shift; + + ## 5' partiality attribute is trusted here !!! + + if ($self->is_5prime_partial()) { + return(); + } + + my @exons = $self->get_exons(); + my $orientation = $self->get_orientation(); + + my @start_codons; + + my $found_cds_flag = 0; + + for (my $i = 0; $i <= $#exons; $i++) { + if (my $cds = $exons[$i]->get_CDS_obj()) { + # found first cds + $found_cds_flag = 1; + my ($cds_end5, $cds_end3) = $cds->get_coords(); + my $cds_len = $cds->length(); + if ($cds_len >= 3) { + ## got start codon in entirety + if ($orientation eq '+') { + push (@start_codons, CDS_exon_obj->new($cds_end5, $cds_end5+2)->set_phase(0)); + last; + } + else { + push (@start_codons, CDS_exon_obj->new($cds_end5, $cds_end5-2)->set_phase(0)); + last; + } + } + else { + ## split start codon + push (@start_codons, $cds); # add current cds as start codon part + my $missing_length = 3 - $cds_len; + + ## examine next cds exon for part of it: + my $next_cds = $exons[$i+1]->get_CDS_obj(); + unless (ref $next_cds) { + die "Error, no next cds for split start codon" . $self->toString(); + } + + my ($next_cds_end5, $next_cds_end3) = $next_cds->get_coords(); + my $next_cds_len = $next_cds->length(); + + if ($next_cds_len >= $missing_length) { + # great, this has everything we need + if ($orientation eq '+') { + push (@start_codons, + CDS_exon_obj->new($next_cds_end5, $next_cds_end5 + $missing_length-1)->set_phase($next_cds->{phase})); + last; + } + else { + push (@start_codons, + CDS_exon_obj->new($next_cds_end5, $next_cds_end5 - $missing_length + 1)->set_phase($next_cds->{phase})); + last; + } + } + else { + ## another split start codon portion. Just add the current cds, and get the first bp from the next cds + push (@start_codons, $next_cds); + + my $final_cds = $exons[$i+2]->get_CDS_obj(); + unless (ref $final_cds) { + die "Error getting final cds of three-part split start codon"; + } + unless ($final_cds->{phase} == 2) { + die "Error, final cds of three-part stop codon is not in phase 2 "; + } + my ($final_cds_end5, $final_cds_end3) = $final_cds->get_coords(); + push (@start_codons, + CDS_exon_obj->new($final_cds_end5, $final_cds_end5)->set_phase(2)); + last; + } + } # end of split start codon + + } # end of found cds + } # end of foreach exon + + unless ($found_cds_flag) { + die "Error, no cds exon found in search of start codon"; + } + + unless (@start_codons) { + die "Error, no start codons found"; + } + ## ensure start codons sum to 3 + my $sum_len = 0; + foreach my $start_codon (@start_codons) { + $sum_len += $start_codon->length(); + } + unless ($sum_len == 3) { + print "Error, sum len of start codons != 3 ( = $sum_len, instead) " . $self->toString() . "starts:\n"; + my $i=0; + foreach my $start (@start_codons) { + $i++; + print "start($i): " . $start->toString(); + } + die; + } + + return (@start_codons); + +} + + + +sub _remove_stop_codons { + my $self = shift; + + ## 3' partiality attribute is trusted here !!! + + if ($self->is_3prime_partial()) { + return (); + } + + my $orientation = $self->get_orientation(); + my @exons = reverse $self->get_exons(); # examining exons in reverse order, starting from stop codon direction. + + my @stop_codons; + + my $found_cds_flag = 0; + + ## find first exon + for (my $i=0; $i <= $#exons; $i++) { + if (my $cds = $exons[$i]->get_CDS_obj()) { + + $found_cds_flag = 1; + + my ($cds_end5, $cds_end3) = $cds->get_coords(); + + my $cds_length = $cds->length(); + if ($cds_length > 3) { + ## cds exon encodes more than just the stop codon + if ($orientation eq '+') { + $cds->{end3} -= 3; + push (@stop_codons, CDS_exon_obj->new($cds_end3 - 2, $cds_end3)->set_phase(0)); + } + else { + $cds->{end3} += 3; + push (@stop_codons, CDS_exon_obj->new($cds_end3 + 2, $cds_end3)->set_phase(0)); + } + last; + + } + elsif ($cds_length == 3) { + ## Just a stop codon exon. We can remove it. + push (@stop_codons, $cds); + $exons[$i]->{CDS_exon_obj} = 0; # nullified + last; + } + + else { + ## cds exon encodes a split stop codon + push (@stop_codons, $cds); # just add the last portion of stop codon + $exons[$i]->{CDS_exon_obj} = 0; # nullified + + ## check next portion of cds exon to see if it contains the rest of the stop + my $next_exon = $exons[$i+1]; + unless (ref $next_exon) { + die "Error, incomplete stop codon and not enough exons! "; + } + my $missing_stop_length = 3 - $cds_length; + my $next_cds_obj = $next_exon->get_CDS_obj(); + unless (ref $next_cds_obj) { + die "Error, next cds obj is missing!"; + } + + my $next_cds_length = $next_cds_obj->length(); + my ($cds_end5, $cds_end3) = $next_cds_obj->get_coords(); + if ($next_cds_length <= $missing_stop_length) { + ## encodes only the second part of the stop codon + # add and nullify + push (@stop_codons, $next_cds_obj); + $next_exon->{CDS_exon_obj} = 0; + + ## get the very last part of the stop + $missing_stop_length -= $next_cds_length; + if ($missing_stop_length > 0) { + ## must be still missing the first bp of the stop codon + if ($missing_stop_length != 1) { + die "Error, too much of the stop codon is left (missing_length = $missing_stop_length). Should only be 1 "; + } + my $next_exon = $exons[$i+2]; + unless (ref $next_exon) { + die "Error, second next exon is unavail "; + } + my $next_cds_obj = $next_exon->get_CDS_obj(); + unless (ref $next_cds_obj) { + die "Error, second next cds obj is unavail"; + } + my $cds_length = $next_cds_obj->length(); + my ($cds_end5, $cds_end3) = $next_cds_obj->get_coords(); + if ($cds_length > 1) { + if ($orientation eq '+') { + $next_cds_obj->{end3}-=1; + push (@stop_codons, CDS_exon_obj->new($cds_end3, $cds_end3)->set_phase(0)); + } + else { + $next_cds_obj->{end3}+=1; + push (@stop_codons, CDS_exon_obj->new($cds_end3, $cds_end3)->set_phase(0)); + } + } + } + } + else { + # split stop codon + #missing length of cds exon is present in the second portion of the stop + if ($orientation eq '+') { + $next_cds_obj->{end3} -= $missing_stop_length; + push (@stop_codons, CDS_exon_obj->new($cds_end3 - $missing_stop_length + 1, $cds_end3)->set_phase(0)); + } + else { + $next_cds_obj->{end3} += $missing_stop_length; + push (@stop_codons, CDS_exon_obj->new($cds_end3 + $missing_stop_length -1, $cds_end3)->set_phase(0)); + } + } + } # end of split stop codon + + last; + + } # end of found cds obj + + + } # end of foreach exon + + unless ($found_cds_flag) { + die "Error, no cds exon was found. "; + } + + + unless (@stop_codons) { + die "Error, no stop codons extracted from non-partial gene."; + } + + @stop_codons = reverse @stop_codons; # reorder according to gene direction + + ## make sure sum (stop_codons) length == 3 + my $sum_len = 0; + foreach my $stop_codon (@stop_codons) { + $sum_len += $stop_codon->length(); + } + if ($sum_len != 3) { + print "Error, stop codons sum length != 3 ( = $sum_len, instead) " . $self->toString(); + my $i=0; + foreach my $stop_codon (@stop_codons) { + print "stop($i): " . $stop_codon->toString(); + } + + die; + } + + return (@stop_codons); + +} + + + +sub has_CDS { + my $self = shift; + + foreach my $exon ($self->get_exons()) { + if (ref ($exon->get_CDS_obj())) { + return (1); + } + } + + return (0); # no cds entry found +} + + +#### +sub set_CDS_phases_from_init_phase { + my ($self, $init_phase) = @_; + + my @exons = $self->get_exons(); + + my $curr_cds_len = $init_phase; + + foreach my $exon (@exons) { + if (my $cds = $exon->get_CDS_obj()) { + $cds->set_phase($curr_cds_len % 3); + my $cds_len = $cds->length(); + $curr_cds_len += $cds_len; + } + } + + return; +} + + + +sub set_CDS_phases { + my ($self, $genomic_seq_ref) = @_; + + + my $start_pos = 1; + if ($self->has_CDS() && ! $self->is_pseudogene()) { + + $self->create_all_sequence_types($genomic_seq_ref); + + my $cds_sequence = $self->get_CDS_sequence(); + my $protein_seq = $self->get_protein_sequence(); + + ## first, clear the partial attributes: + $self->set_5prime_partial(0); + $self->set_3prime_partial(0); + + + if ($protein_seq !~ /^M/) { + # lacks start codon + $self->set_5prime_partial(1); + } + if ($protein_seq !~ /\*$/) { + # lacks stop codon + $self->set_3prime_partial(1); + } + + $start_pos = $self->_get_cds_start_pos($cds_sequence); + + ## must set phase based on codon start position: + ## my definition of phase here is the actual codon position of the first base in the CDS sequence. + ## (note this differs from the GFF3 spec, and is adjusted for in the to_GFF3_format() method. + ## + + my $first_phase; + if ($start_pos == 0) { + # ATG XXX ... + # 012 012 012 + + $first_phase = 0; + } + elsif ($start_pos == 1) { + # XAT GXX ... + # 201 201 201 + + $first_phase = 2; + } + elsif ($start_pos == 2) { + # XXA TGX ... + # 120 120 120 + + $first_phase = 1 + } + else { + confess "Error, start pos: $start_pos doesn't make sense here... must be a bug."; + } + + my @exons = $self->get_exons(); + my @cds_objs; + foreach my $exon (@exons) { + my $cds = $exon->get_CDS_obj(); + if (ref $cds) { + push (@cds_objs, $cds); + } + } + + my $cds_obj = shift @cds_objs; + $cds_obj->{phase} = $first_phase; + my $cds_length = abs ($cds_obj->{end3} - $cds_obj->{end5}) + 1; + + if ($first_phase != 0) { + # and now I understand why the GFF3 phase definition differs from mine. :-) + if ($first_phase == 1) { + $cds_length -= 2; + } + elsif ($first_phase == 2) { + $cds_length -= 1; + } + else { + confess "Error, first phase set to: $first_phase, which is nonsensical"; + } + } + + while (@cds_objs) { + my $next_cds_obj = shift @cds_objs; + $next_cds_obj->{phase} = $cds_length % 3; + $cds_length += abs ($next_cds_obj->{end3} - $next_cds_obj->{end5}) + 1; + } + } + + foreach my $isoform ($self->get_additional_isoforms()) { + $isoform->set_CDS_phases($genomic_seq_ref); + } + + return; + +} + + +sub get_first_CDS_segment { + my $gene_obj = shift; + my @exons = $gene_obj->get_exons(); + + foreach my $exon (@exons) { + if (my $cds = $exon->get_CDS_exon_obj()) { + return ($cds); + } + } + + return undef; +} + +sub _get_cds_start_pos { + my ($self, $cds_sequence) = @_; + my $cds_length = length($cds_sequence); + # if cds is set of triplets, assume translate at codon pos 1. + my $codon_start; + + ## must determine where translation starts: + my $new_orfFinder = new Longest_orf(); + $new_orfFinder->allow_partials(); + $new_orfFinder->forward_strand_only(); + + my $longest_orf = $new_orfFinder->get_longest_orf($cds_sequence); + + unless (ref $longest_orf) { + die "No longest ORF found in sequence"; + } + + ## examine the first three ORFs, prefer long orf with stop codon. + my $orfPos = $longest_orf->{start}; #init to first, longest orf. + unless (defined $orfPos) { + die "Error, orfPos not defined! " . Dumper ($longest_orf); + } + + my $bestOrfPos; + my @allOrfs = $new_orfFinder->orfs(); + + for my $orfIndex (0..2) { + my $orf = $allOrfs[$orfIndex]; + if ($orf) { + my $start = $orf->{start}; + my $length = $orf->{length}; + my $protein = $orf->{protein}; + if ($length > $cds_length - 3 && $start <= 3 && $protein =~ /\*$/) { + unless ($bestOrfPos) { + $bestOrfPos = $start; + } + } + } + } + + if ($bestOrfPos && $bestOrfPos != $orfPos) { + $orfPos = $bestOrfPos; + } + + if ($orfPos >3) { + confess "Error, longest ORF is found at position $orfPos, and should be between 1 and 3. What's wrong with your gene?" . $self->toString(); + } + $codon_start = $orfPos; + + #longest orf apparently using 1-based rather than 0-based coordinates. + + $codon_start -= 1; + + return ($codon_start); +} + + +=over 4 + +=item dispose() + +B Sets all attributes = 0, hopefully to faciliate targeting for garbage collection. (experimental method) + +B none + +B none + +=back + +=cut + +sub dispose { + my $self = shift; + foreach my $att (keys %$self) { + $self->{$att} = 0; + } +} + + + +sub DESTROY { + my $self = shift; + + warn "DESTROYING gene_obj: " . $self->{TU_feat_name} . "," . $self->{Model_feat_name} . "\n" if $main::DEBUG; + +} + + +sub validate_start_codon { + ## requires that you have the CDS sequence already set + my $self = shift; + + my $cds_sequence = $self->get_CDS_sequence() or confess "Error, cannot get CDS sequence. It must be built prior to calling this method"; + ## currently, only trust Met start codons. + my $start_codon = uc substr($cds_sequence, 0, 3); + if ($start_codon ne "ATG") { + die $self->toString() . "Error, start codon is not M (codon $start_codon instead)!"; + # call within an eval block to catch exception + } +} + + +sub validate_stop_codon { + ## requires that you have the CDS sequence already set + my $self = shift; + + my $cds_sequence = $self->get_CDS_sequence() or confess "Error, cannot get CDS sequence. It must be built prior to calling this method"; + + my @stop_codons = &Nuc_translator::get_stop_codons(); + + my $curr_stop_codon = substr($cds_sequence, length($cds_sequence)-3, 3); + + my $found_stop_codon_flag = 0; + foreach my $stop (@stop_codons) { + if ($stop eq $curr_stop_codon) { + $found_stop_codon_flag = 1; + last; + } + } + + unless ($found_stop_codon_flag) { + die $self->toString() . "Error, stop codon $curr_stop_codon is not an acceptable stop codon: [@stop_codons]\n"; + } + +} + + + + +###################################################################################################################################### +###################################################################################################################################### + + +=head1 NAME + +package mRNA_exon_obj + +=cut + +=head1 DESCRIPTION + + The mRNA_exon_obj represents an individual spliced mRNA exon of a gene. The coordinates of the exon can be manipulated, and the mRNA_exon_obj can contain a single CDS_exon_obj. A mRNA_exon_obj lacking a CDS_exon_obj component is an untranslated (UTR) exon. + + A mature Gene_obj is expected to have at least one mRNA_exon_obj component. + +=cut + + +package mRNA_exon_obj; + +use strict; +use warnings; +use Storable qw (store retrieve freeze thaw dclone); + +=over 4 + +=item new() + +B Instantiates an mRNA_exon_obj + +B <(end5, end3)> + +The end5 and end3 coordinates can be optionally passed into the constructor to set these attributes. Alternatively, the set_coords() method can be used to set these values. + +B $mRNA_exon_obj + +=back + +=cut + + + ; + +sub new { + shift; + my $self = { end5 => 0, # stores end5 of mRNA exon + end3 => 0, # stores end3 of mRNA exon + CDS_exon_obj => 0, # stores object reference to CDS_obj + feat_name => 0, # stores TIGR temp id + strand => undef, # +|- + }; + + # end5 and end3 can be included as parameters in constructor. + if (@_) { + my ($end5, $end3) = @_; + if (defined($end5) && defined($end3)) { + $self->{end5} = $end5; + $self->{end3} = $end3; + } + } + + bless ($self); + return ($self); +} + + + +=over 4 + +=item get_CDS_obj() + +B Retrieves the CDS_exon_obj component of this mRNA_exon_obj + +B none + +B $cds_exon_obj + +If no CDS_exon_obj is attached, returns 0 + +=back + +=cut + + ; + +sub get_CDS_obj { + my $self = shift; + return ($self->{CDS_exon_obj}); +} + + +## alias +sub get_CDS_exon_obj { + my $self = shift; + return ($self->get_CDS_obj()); +} + + +=over 4 + +=item get_mRNA_exon_end5_end3() + +B Retrieves the end5, end3 coordinates of the exon + +**Method Deprecated**, use get_coords() + +B none + +B (end5, end3) + +=back + +=cut + + +sub get_mRNA_exon_end5_end3 { + my $self = shift; + return ($self->{end5}, $self->{end3}); +} + + + +=over 4 + +=item set_CDS_exon_obj() + +B Sets the CDS_exon_obj of the mRNA_exon_obj + +B $cds_exon_obj + +B none + +=back + +=cut + + ; +sub set_CDS_exon_obj { + my $self = shift; + my $ref = shift; + if (ref($ref)) { + $self->{CDS_exon_obj} = $ref; + } +} + + + +#### +sub delete_CDS_exon_obj { + my $self = shift; + $self->{CDS_exon_obj} = undef; + return; +} + + +=over 4 + +=item add_CDS_exon_obj() + +B Instantiates and adds a new CDS_exon_obj to the mRNA_exon_obj given the CDS coordinates. + +B (end5, end3) + +B none + +=back + +=cut + + +sub add_CDS_exon_obj { + my $self = shift; + my ($end5, $end3) = @_; + my $cds_obj = CDS_exon_obj->new ($end5, $end3); + $self->set_CDS_exon_obj($cds_obj); +} + + +=over 4 + +=item set_feat_name() + +B Sets the feat_name attribute of the mRNA_exon_obj + +B $feat_name + +B none + +=back + +=cut + + + +sub set_feat_name { + my $self = shift; + my $feat_name = shift; + $self->{feat_name} = $feat_name; +} + + +=over 4 + +=item clone_exon() + +B Creates a deep clone of this mRNA_exon_obj, using dclone() of Storable.pm + +B none + +B $mRNA_exon_obj + +=back + +=cut + + + +sub clone_exon { + my $self = shift; + + my $clone_exon = dclone($self); + + return ($clone_exon); +} + + + +=over 4 + +=item get_CDS_end5_end3 () + +B Retrieves end5, end3 of the CDS_exon_obj component of this mRNA_exon_obj + +B none + +B (end5, end3) + +An empty array is returned if no CDS_exon_obj is attached. + +=back + +=cut + + +sub get_CDS_end5_end3 { + my $self = shift; + my $cds_obj = $self->get_CDS_obj(); + if ($cds_obj) { + return ($cds_obj->get_CDS_end5_end3()); + } else { + return ( () ); + } +} + + +=over 4 + +=item get_coords() + +B Retrieves the end5, end3 coordinates of this mRNA_exon_obj + +B none + +B (end5, end3) + +=back + +=cut + + +sub get_coords { + my $self = shift; + return ($self->get_mRNA_exon_end5_end3()); +} + + +=over 4 + +=item set_coords() + +B Sets the end5, end3 coordinates of the mRNA_exon_obj + +B (end5, end3) + +B none + +=back + +=cut + + +## simpler coord setting (end5, end3) +sub set_coords { + my $self = shift; + my $end5 = shift; + my $end3 = shift; + $self->{end5} = $end5; + $self->{end3} = $end3; +} + + +=over 4 + +=item get_strand() + +B Retrieves the orientation of the mRNA_exon_obj based on gene models transcribed orientation. + +B none + +B +|-|undef + +If end5 == end3, strand orientation cannot be inferred based on coordinates alone, so undef is returned. + +=back + +=cut + + + ; + +sub get_orientation { + # determine positive or reverse orientation + my $self = shift; + return ($self->{strand}); +} + + +sub get_strand { ## preferred + my $self = shift; + return($self->get_orientation()); +} + + +#### +sub merge_exon { + my $self = shift; + my $other_exon = shift; + + my $cds = $self->get_CDS_exon_obj(); + + my $other_cds = $other_exon->get_CDS_exon_obj(); + + if ($other_cds) { + if ($cds) { + $cds->merge_CDS($other_cds); + } + else { + # current exon lacks cds. Set this one to it. + $self->set_CDS_exon_obj($other_cds); + } + } + + + ## merge the exons. + my @coords = sort {$a<=>$b} ($self->get_coords(), $other_exon->get_coords()); + my $lend = shift @coords; + my $rend = pop @coords; + + my ($new_end5, $new_end3) = ($self->get_orientation() eq '+') ? ($lend, $rend) : ($rend, $lend); + + $self->set_coords($new_end5, $new_end3); + + return; +} + + + + + + +=over 4 + +=item toString() + +B Provides a textual description of the mRNA_exon_obj + +B none + +B $text + +=back + +=cut + + ; + + +sub toString { + my $self = shift; + my @coords = $self->get_mRNA_exon_end5_end3(); + my $feat_name = $self->{feat_name}; + my $text = ""; + if ($feat_name) { + $text .= "feat_name: $feat_name\t"; + } + $text .= "end5 " . $coords[0] . "\tend3 " . $coords[1] . "\n"; + return ($text); +} + + +sub length { + my $self = shift; + + my $len = abs ($self->{end5} - $self->{end3}) + 1; + + return($len); +} + + + + + +########################################################################################################################## +########################################################################################################################## + + + +=head1 NAME + +package CDS_exon_obj + +=cut + + +=head1 DESCRIPTION + + The CDS_exon_obj represents the protein-coding portion of an mRNA_exon_obj. + +=cut + + + +package CDS_exon_obj; + +use strict; +use warnings; +use Storable qw (store retrieve freeze thaw dclone); +use Carp; + + +=over 4 + +=item new() + +B Cosntructor for the CDS_exon_obj + +B <(end5, end3)> + +The (end5, end3) parameter is optional. Alternatively, the set_coords() method can be used to set these values. + +B $cds_exon_obj + +=back + +=cut + + ; + +sub new { + shift; + my $self = { end5 => 0, #stores end5 of cds exon + end3 => 0, #stores end3 of cds exon + phase => undef, #must set if to output in gff3 format. + feat_name => 0, #tigr's temp id + strand => undef, # +|- + }; + + + # end5 and end3 are allowed constructor parameters + if (@_) { + my ($end5, $end3) = @_; + if (defined ($end5) && defined ($end3)) { + $self->{end5} = $end5; + $self->{end3} = $end3; + } + } + bless ($self); + return ($self); +} + + + +=over 4 + +=item set_feat_name() + +B Sets the feat_name attribute value of the CDS_exon_obj + +B $feat_name + +B none + +=back + +=cut + + +sub set_feat_name { + my $self = shift; + my $feat_name = shift; + $self->{feat_name} = $feat_name; +} + + +=over 4 + +=item get_CDS_end5_end3() + +B Retrieves the end5, end3 coordinates of the CDS_exon_obj + +** Method deprecated **, use get_coords() + + +B none + +B (end5, end3) + +=back + +=cut + + +sub get_CDS_end5_end3 { + my $self = shift; + return ($self->{end5}, $self->{end3}); +} + + + +=over 4 + +=item set_coords() + +B Sets the (end5, end3) values of the CDS_exon_obj + +B (end5, end3) + +B none + +=back + +=cut + + + +sub set_coords { + my $self = shift; + my $end5 = shift; + my $end3 = shift; + $self->{end5} = $end5; + $self->{end3} = $end3; +} + +=over 4 + +=item get_coords() + +B Retrieves the (end5, end3) coordinates of the CDS_exon_obj + +B none + +B (end5, end3) + + +The get_coords() method behaves similarly among Gene_obj, mRNA_exon_obj, and CDS_exon_obj, and is generally preferred to other existing methods for extracting these coordinate values. Other methods persist for backwards compatibility with older applications, but have been largely deprecated. + + +=back + +=cut + + + +sub get_coords { + my $self = shift; + return ($self->get_CDS_end5_end3()); +} + + +=over 4 + +=item get_orientation() + +B Retrieves the orientation of the CDS_exon_obj based on gene models orientation. + +B none + +B +|-|undef + +undef returned if end5 == end3 + +=back + +=cut + + ; + +sub get_orientation { + # determine positive or reverse orientation + my $self = shift; + return ($self->{strand}); +} + + +sub get_strand { ## preferred + my $self = shift; + return($self->get_orientation()); +} + + +=over 4 + +=item toString() + +B Retrieves a textual description of the CDS_exon_obj + +B none + +B $text + +=back + +=cut + + + +=over 4 + +=item clone_cds() + +B Creates a deep clone of this CDS_exon_obj, using dclone() of Storable.pm + +B none + +B $mRNA_exon_obj + +=back + +=cut + + + +sub clone_cds { + my $self = shift; + + my $clone_cds = dclone($self); + + return ($clone_cds); +} + + +=over 4 + +=item length() + +B length of this cds segment + +B none + +B int + +=back + +=cut + + +sub length { + my $self = shift; + my $length = abs ($self->{end3} - $self->{end5}) + 1; + return ($length); +} + + + +=over 4 + +=item set_phase() + +B set phase of the CDS incident bp + +B [012] + +B self + + +phase 0 = first bp of codon +phase 1 = second bp of codon +phase 2 = third bp of codon + + +=back + +=cut + + +sub set_phase { + my $self = shift; + my $phase = shift; + $self->{phase} = $phase; + return($self); +} + +=over 4 + +=item get_phase() + +B gets phase of the CDS incident bp + +B none + +B [012] or undef if not set + + +phase 0 = first bp of codon +phase 1 = second bp of codon +phase 2 = third bp of codon + + +=back + +=cut + + + + +sub get_phase { + my $self = shift; + my $phase = $self->{phase}; + return($phase); +} + + + +#### +sub merge_CDS { + my $self = shift; + my $other_cds = shift; + + my $orientation = $self->get_orientation(); + unless ($orientation) { + confess "Error, self CDS lacks orientation\n"; + } + + my @coords = sort {$a<=>$b} ($self->get_coords(), $other_cds->get_coords()); + my $lend = shift @coords; + my $rend = pop @coords; + + unless ($lend && $rend) { + confess "Error, trying to merge CDSs but coordinates are not available: \n" + . "self: " . $self->toString() + . "\n" + . "other: " . $other_cds->toString() . "\n"; + } + + my ($end5, $end3) = ($orientation eq '+') ? ($lend, $rend) : ($rend, $lend); + + $self->set_coords($end5, $end3); +} + +sub toString { + my $self = shift; + my @coords = $self->get_CDS_end5_end3(); + my $feat_name = $self->{feat_name}; + my $text = ""; + if ($feat_name) { + $text .= "feat_name: $feat_name\t"; + } + $text .= "end5 " . $coords[0] . "\tend3 " . $coords[1] . "\n"; + return ($text); +} + + +1; + + + + + + + + + + + + + + diff --git a/99.scripts/trinity_utils/PerlLib/Gene_obj_indexer.pm b/99.scripts/trinity_utils/PerlLib/Gene_obj_indexer.pm new file mode 100644 index 0000000..c44f78b --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Gene_obj_indexer.pm @@ -0,0 +1,75 @@ +#!/usr/local/bin/perl + +package Gene_obj_indexer; +use strict; +use warnings; +use base qw(TiedHash); +use Gene_obj; +use Storable qw (thaw nfreeze); +use Carp; + + +#### +sub new { + my $packagename = shift; + + my $self = $packagename->SUPER::new(@_); + + return ($self); + +} + +#### +sub store_gene { + my ($self, $identifier, $gene_obj) = @_; + + + unless (ref $gene_obj) { + confess "Error, no gene_obj as param"; + } + + my $blob = nfreeze ($gene_obj); + + my $success = 0; + + while (! $success) { + $self->store_key_value($identifier, $blob); + + eval { + my $gene_obj = $self->get_gene($identifier); + + }; + if ($@) { + warn "error trying to store gene $identifier using berkeley db. Trying again...\n"; + } + else { + # worked. + $success = 1; + } + } + +} + + +#### +sub get_gene { + my $self = shift; + my $identifier = shift; + + my $blob = $self->get_value($identifier); + + unless ($blob) { + confess "Error, no gene obj retrieved based on identifier $identifier"; + } + + my $gene_obj = thaw($blob); + unless (ref $gene_obj) { + confess "Error retrieving gene_obj based on identifier $identifier. Data retrieved but not thawed properly.\n"; + } + + return ($gene_obj); +} + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/AlignGraph.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/AlignGraph.pm new file mode 100644 index 0000000..fc77949 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/AlignGraph.pm @@ -0,0 +1,149 @@ +package main; +our $SEE; + +package AlignGraph; + +use strict; +use warnings; +use Carp; +use AlignNode; + +use base qw (ReadCoverageGraph); + +no warnings qw (recursion); + + +sub new { + my $packagename = shift; + + my $self = $packagename->SUPER::new(); + + + ## Nodes exist in order of end5 -> end3 at all times. Strand is included in the node name just for safety reasons. + + bless ($self, $packagename); + + return($self); +} + + +sub add_alignment { + my $self = shift; + + my ($read_acc, $scaffold, $strand, $genome_coords_aref) = @_; + + my @align_positions; + + foreach my $coordset (sort {$a->[0]<=>$b->[0]} @$genome_coords_aref) { + + my ($lend, $rend) = sort {$a<=>$b} @$coordset; + + for (my $i = $lend; $i <= $rend; $i++) { + push (@align_positions, $i); + } + } + + $self->_add_ordered_positions_to_graph(\@align_positions, $strand, $read_acc); + + return; + +} + + + +sub _add_ordered_positions_to_graph { + my $self = shift; + my ($ordered_positions_aref, $strand, $read_acc) = @_; + + unless (ref $ordered_positions_aref eq 'ARRAY') { + confess "Error, require ordered position list"; + } + unless ($strand =~ /^[\+\-]$/) { + confess "strand must be: + or - "; + } + + my @align_positions = @$ordered_positions_aref; + + if ($strand eq '-') { + @align_positions = reverse @align_positions; + } + + my $prev_align_pos = shift @align_positions; + while (@align_positions) { + my $next_align_pos = shift @align_positions; + + # print "\tadding $next_align_pos\n"; + + my $prev_align_node = $self->get_or_create_node("$prev_align_pos,$strand", $read_acc); + my $next_align_node = $self->get_or_create_node("$next_align_pos,$strand", $read_acc); + + $self->link_adjacent_nodes($prev_align_node, $next_align_node); + + $prev_align_pos = $next_align_pos; + + } + + return; +} + +sub get_all_nodes { + my $self = shift; + + return( sort {$a->{_value} cmp $b->{_value}} $self->SUPER::get_all_nodes()); + +} + + + +1; #EOM + + +=CIGAR_format + +from: http://bioperl.org/pipermail/bioperl-l/2003-March/011591.html + +cigar line format (where CIGAR stands for Concise +Idiosyncratic Gapped Alignment Report). + +In the cigar line format alignments are sotred as follows: + +M: Match +D: Deletino +I: Insertion + +An example of an alignment for a hypthetical protein match is shown +below: + + +Query: 42 PGPAGLP----GSVGLQGPRGLRGPLP-GPLGPPL... + PG P G GP R PLGP +Sbjct: 1672 PGTP*TPLVPLGPWVPLGPSSPR--LPSGPLGPTD... + + +protein_align_feature table as the following cigar line: + +7M4D12M2I2MD7M + + + + From SAM documentation: + +Clipped alignment. In Smith-Waterman alignment, a sequence may not be aligned from the first residue to the last one. +Subsequences at the ends may be clipped off. We introduce operation ʻSʼ to describe (softly) clipped alignment. Here is +an example. Suppose the clipped alignment is: + +REF: AGCTAGCATCGTGTCGCCCGTCTAGCATACGCATGATCGACTGTCAGCTAGTCAGACTAGTCGATCGATGTG +READ: gggGTGTAACC-GACTAGgggg + +where on the read sequence, bases in uppercase are matches and bases in lowercase are clipped off. The CIGAR for +this alignment is: 3S8M1D6M4S. +Spliced alignment. In cDNA-to-genome alignment, we may want to distinguish introns from deletions in exons. We +introduce operation ʻNʼ to represent long skip on the reference sequence. Suppose the spliced alignment is: + +REF: AGCTAGCATCGTGTCGCCCGTCTAGCATACGCATGATCGACTGTCAGCTAGTCAGACTAGTCGATCGATGTG +READ: GTGTAACCC................................TCAGAATA + +where ʻ...ʼ on the read sequence indicates the intron. The CIGAR for this alignment is: 9M32N8M. + +=cut + diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/AlignNode.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/AlignNode.pm new file mode 100644 index 0000000..6e984d9 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/AlignNode.pm @@ -0,0 +1,19 @@ +package AlignNode; + +use base qw (ReadCoverageNode); + +## Instead of storing Kmers, will store base number and strand positions. + +sub new { + my $packagename = shift; + my ($stranded_base, $read_accession) = @_; + + my $self = $packagename->SUPER::new($stranded_base, $read_accession); + + bless ($self, $packagename); + + return($self); +} + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/GenericGraph.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/GenericGraph.pm new file mode 100644 index 0000000..5bf298f --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/GenericGraph.pm @@ -0,0 +1,221 @@ +package GenericGraph; + +use strict; +use warnings; +use Carp; + +sub new { + my $packagename = shift; + + my $self = { + _nodes => {}, # node_name => node_reference + _edge_counter => {}, # prev_node -> after_node = count + }; + + bless ($self, $packagename); + + return($self); +} + + +sub get_or_create_node { + my $self = shift; + my $node_name = shift; + + if ($self->node_exists($node_name)) { + return($self->get_node($node_name)); + } + else { + # instantiate it, add it to the graph + return($self->create_node($node_name)); + } +} + + +sub node_exists { + my $self = shift; + my $node_name = shift; + + unless ($node_name =~ /\w/) { + confess "Error, node_name required"; + } + + if (exists $self->{_nodes}->{$node_name}) { + return(1); + } + else { + return(0); + } +} + +sub get_node { + my $self = shift; + my $node_name = shift; + + if (! $self->node_exists($node_name)) { + confess "Error, $node_name doesn't exist in graph"; + } + + my $node = $self->{_nodes}->{$node_name}; + + return($node); +} + +sub get_all_nodes { + my $self = shift; + + return(values %{$self->{_nodes}}); +} + + +sub create_node { + my $self = shift; + + my $node_name = shift; + + if ($self->node_exists($node_name)) { + confess "Error, node $node_name already exists in the graph"; + } + + my $node = new GenericNode($node_name); + + $self->{_nodes}->{$node_name} = $node; + + return($node); +} + + +sub link_adjacent_nodes { + my $self = shift; + my ($before_node, $after_node, $edge_increment) = @_; + + unless (ref $before_node && ref $after_node) { + confess "Error, need both before and after nodes for linking"; + } + + $before_node->add_next_node($after_node); + $after_node->add_prev_node($before_node); + + if (defined($edge_increment)) { + if ($edge_increment =~ /^\d+/ && $edge_increment > 0) { + $self->{_edge_counter}->{$before_node}->{$after_node} += $edge_increment; + } + } + else { + $self->{_edge_counter}->{$before_node}->{$after_node}++; + } + + + return; +} + + +sub get_edge_count { + my $self = shift; + my ($prev_node, $next_node) = @_; + + my $edge_count = $self->{_edge_counter}->{$prev_node}->{$next_node} || 0; + + return($edge_count); +} + + +#### +sub prune_nodes_from_graph { + my $self = shift; + my @nodes = @_; + + my $graph_nodes_href = $self->{_nodes}; + + foreach my $node (@nodes) { + + delete ($self->{_edge_counter}->{$node}); # remove edge counts starting at current node. + + my $node_name = $node->get_value(); + delete $graph_nodes_href->{$node_name}; + + my @next_nodes = $node->get_all_next_nodes(); + my @prev_nodes = $node->get_all_prev_nodes(); + + foreach my $prev_node (@prev_nodes) { + $prev_node->delete_next_node($node); + $node->delete_prev_node($prev_node); + + delete ($self->{_edge_counter}->{$prev_node}->{$node}); # remove edge counts for prev nodes linking to current node. + + } + + foreach my $next_node (@next_nodes) { + + $next_node->delete_prev_node($node); + $node->delete_next_node($next_node); + } + + + } + + + return; +} + +#### +sub prune_edge { + my $self = shift; + my ($prev_node, $node) = @_; + + # Sever connection between nodes. + + $prev_node->delete_next_node($node); + $node->delete_prev_node($prev_node); + + delete ($self->{_edge_counter}->{$prev_node}->{$node}); + + return; +} + + +#### +sub print_path { + my @nodes = @_; + + my $counter = 0; + foreach my $node (@nodes) { + $counter++; + printf("%4s", $counter); + print " " . $node->get_value() . "\n"; + } + print "\n"; + + return; +} + +#### +sub toString { + my $self = shift; + + my @nodes = $self->get_all_nodes(); + + my $text = ""; + foreach my $node (@nodes) { + $text .= $node->toString() . "\n"; + } + + return($text); +} + +#### +sub get_root_nodes { + my $self = shift; + + my @roots; + foreach my $node ($self->get_all_nodes()) { + unless ($node->get_all_prev_nodes()) { + push (@roots, $node); + } + } + + return(@roots); +} + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/GenericNode.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/GenericNode.pm new file mode 100644 index 0000000..1c9fa22 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/GenericNode.pm @@ -0,0 +1,170 @@ +package GenericNode; + +use strict; +use warnings; + +use Carp; + +my $ID_counter = 0; + +sub new { + my $packagename = shift; + + my ($value) = @_; + + my $self = { + + _value => $value, + _prev => {}, + _next => {}, + + _ID => ++$ID_counter, + + _colors => {}, + + }; + + bless ($self, $packagename); + + return($self); + +} + +sub get_hexID { + my $self = shift; + + my $id = sprintf("%x", $self->{_ID}); + + return("$id"); +} + +sub get_ID { + my $self = shift; + return($self->{_ID}); +} + + +sub get_value { + my $self = shift; + + return($self->{_value}); +} + +sub set_value { + my ($self) = shift; + my ($value) = @_; + + $self->{_value} = $value; +} + +sub add_prev_node { + my $self = shift; + my ($prev_node) = @_; + + unless (ref $prev_node) { + croak "Error, prev_node should be a node object"; + } + $self->{_prev}->{$prev_node} = $prev_node; + return; +} + +sub add_next_node { + my $self = shift; + my ($next_node) = @_; + + unless (ref $next_node) { + croak "Error, next_node should be a node object"; + } + + $self->{_next}->{$next_node} = $next_node; +} + +sub has_next_node { + my $self = shift; + my $next_node = shift; + + if (exists $self->{_next}->{$next_node}) { + return(1); + } + else { + return(0); + } +} + +sub has_prev_node { + my $self = shift; + my $prev_node = shift; + + if (exists $self->{_prev}->{$prev_node}) { + return(1); + } + else { + return(0); + } +} + + + +sub get_all_prev_nodes { + my $self = shift; + my @prev_nodes = grep { ref($_) } values %{$self->{_prev}}; + + #print "Prev: " . join(",", @prev_nodes) . "\n"; + + return(@prev_nodes); + +} + +sub get_all_next_nodes { + my $self = shift; + my @next_nodes = grep { ref($_) } values %{$self->{_next}}; + + #print "Next: " . join(",", @next_nodes) . "\n"; + + return(@next_nodes); + +} + +sub delete_prev_node { + my $self = shift; + my $node = shift; + + my $prev_nodes_href = $self->{_prev}; + delete $prev_nodes_href->{$node}; + + return; +} + +sub delete_next_node { + my $self = shift; + my $node = shift; + + my $next_nodes_href = $self->{_next}; + delete $next_nodes_href->{$node}; + + return; +} + + +sub add_color { + my $self = shift; + + my $color = shift; + + $self->{_colors}->{$color}++; + + return; +} + +sub get_colors { + my $self = shift; + + return(keys %{$self->{_colors}}); +} + + + + +1; #EOM + + diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/KmerGraph.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/KmerGraph.pm new file mode 100644 index 0000000..2953147 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/KmerGraph.pm @@ -0,0 +1,1134 @@ +package main; +our $SEE; + +package KmerGraph; + +use strict; +use warnings; +use Carp; +use KmerNode; +use ReadTracker; + +use base qw (ReadCoverageGraph); + +no warnings qw (recursion); + + +my $MIN_DISPLAY_SEQ_TAG_LENGTH = 10; # lower than this, and shows up in the dot file as a label. + +sub new { + my $packagename = shift; + my $KmerLength = shift; + + unless ($KmerLength > 1) { croak "Error, need KmerLength > 1"; } + + my $self = { + _nodes => {}, #kmerSeq => nodeAddress until after compaction... becomes nodeaddress => $nodeaddress + KmerLength => $KmerLength, + compacted => 0, + color_to_accs => {}, + + }; + + bless ($self, $packagename); + + return($self); +} + + +sub add_sequence_to_graph { + my $self = shift; + my ($acc, $sequence, $weight, $color) = @_; ## Sequence type can be undef, ReferenceTranscript, or GreedyPath + + unless (defined($weight) && $weight =~ /^\d+/) { + confess "Error, params: (seq, weight, color), and weight must be numeric."; + } + + if ($self->{compacted}) { + confess "Cannot add more sequences to the graph after it has been compacted"; + } + + push (@{$self->{color_to_accs}->{$color}}, $acc); # add this accession to color indicator. # prefer 1-1 relationship here. + + $sequence = uc $sequence; + + my $KmerLength = $self->{KmerLength}; + + if (length($sequence) < $KmerLength+1) { + warn "sequence $sequence is less than KmerLength $KmerLength+1; skipping it\n"; + return; + } + + my $prevKmer = substr($sequence, 0, $KmerLength); + my $prevKmerNode = $self->get_or_create_node($prevKmer); + $prevKmerNode->{_count}++; + $prevKmerNode->track_reads($acc); + + if ($color) { + $prevKmerNode->add_color($color); + } + + for (my $i = 1; $i <= length($sequence)-$KmerLength; $i++) { + my $nextKmer = substr($sequence, $i, $KmerLength); + my $nextKmerNode = $self->get_or_create_node($nextKmer); + $nextKmerNode->track_reads($acc); + + if ($color) { + $nextKmerNode->add_color($color); + } + + $nextKmerNode->{_count}++; + + my $prev_edge_weight = $self->get_edge_count($prevKmerNode, $nextKmerNode); + + ## only link together kmers that contain only recognizable nucleotides + $self->link_adjacent_nodes($prevKmerNode, $nextKmerNode, $weight); + + my $edge_weight = $self->get_edge_count($prevKmerNode, $nextKmerNode); + if ($edge_weight < $weight) { + confess "Error, added weight $weight to edge $prevKmerNode -> $nextKmerNode and didn't stick"; + } + + if ($prev_edge_weight > $edge_weight) { + confess "Error, after adding new edge, prev weight of $prev_edge_weight became: $edge_weight "; + } + #print "$prevKmerNode -> $nextKmerNode : $prev_edge_weight -> $edge_weight\n"; + + $prevKmerNode = $nextKmerNode; + $prevKmer = $nextKmer; + } + + return; +} + +sub get_or_create_node { + my $self = shift; + my ($node_name, $read_accession) = @_; + + if ($self->node_exists($node_name)) { + my $node = $self->get_node($node_name); + return($node); + + } + else { + # instantiate it, add it to the graph + return($self->create_node($node_name)); + } +} + + +sub create_node { + my $self = shift; + + my ($node_name) = @_; + + if ($self->node_exists($node_name)) { + confess "Error, node $node_name already exists in the graph"; + } + + my $node = new KmerNode($node_name); + + $self->{_nodes}->{$node_name} = $node; + + return($node); +} + + + + +#### +sub compact_graph { + my $self = shift; + + $self->validate_graph(); + + my $kmer_length = $self->{KmerLength}; + + unless ($self->{compacted}) { + $self->_prep_graph_for_compaction(); + + $self->{compacted} = 1; + } + + my $COMPACTED = 0; # flag + + + ## identify all nodes that are branched on the left and not branched on the right + ## then join together the unbranched nodes starting from these and walking right. + + my @nodes = $self->get_all_nodes(); + + ## merge unbranched nodes in runs. + + + + + #### + #### \ + #### (-)--- + #### / + + # or + + # (-)--- + + + # or + + #### (-)------------ + #### / + #### - + #### \ + #### (-) ------------ + + + # or, single line with incompatible decorations: + + # aaaaaaaaaa b bbbbbbbbbb + # ----------(-)---------- + + + my @init_nodes; + foreach my $node (@nodes) { + my @prev_nodes = $node->get_all_prev_nodes(); + + my @next_nodes = $node->get_all_next_nodes(); + + if + + ( (scalar(@next_nodes) == 1) + && + + ( + + + + + ( + (scalar(@prev_nodes) != 1) # branched before or no node before. + || + (scalar (@prev_nodes) == 1 && scalar($prev_nodes[0]->get_all_next_nodes() > 1) ) # previous node branches off. + + ) + + || + + ## decoration switch + (scalar(@prev_nodes) == 1 && (! &_compatible_decorations($node, $prev_nodes[0])) && (&_compatible_decorations($node, $next_nodes[0])) ) + + ) + + ) + { + # got one + push (@init_nodes, $node); + } + + } + + print "Got " . scalar(@init_nodes) . " init nodes.\n" if $main::SEE; + + #foreach my $node (@init_nodes) { + # print "Init node $node:\n" . $node->toString() . "\n";# if $main::SEE; + #} + + + ## Collapse neighboring nodes before branching + foreach my $node (@init_nodes) { + + #print "Init node $node:\n" . $node->toString() . "\n";# if $main::SEE; + + my @unbranched_path; + + my ($next_node) = $node->get_all_next_nodes(); + + unless ($next_node) { + print "warning, init node now lacks next nodes... skipping.\n"; # potential bug? or circular dealt with already? + next; + } + + my %seen = ($node => 1); + + while (1) { + # avoid circularity + if ($seen{$next_node}) { + print "Seen $next_node already... avoiding circularity.\n" if $main::SEE; + last; + } + $seen{$next_node} = 1; + + unless (&_compatible_decorations($node, $next_node)) { + print "Incompatible decorations.\n" if $main::SEE; + last; + } + + # check for left branching on next node + my $num_prev_nodes = scalar ($next_node->get_all_prev_nodes()); + if (scalar $num_prev_nodes > 1) { + print "Got $num_prev_nodes prev nodes for $next_node; terminating extension.\n" if $main::SEE; + last; + } + + push (@unbranched_path, $next_node); + my @next_nodes = $next_node->get_all_next_nodes(); + + if (scalar @next_nodes != 1) { + last; + } + $next_node = shift @next_nodes; + + } + + if ($main::SEE) { + print "Single unbranched path:\n"; + $self->print_path($node, @unbranched_path); + } + + foreach my $next_node (@unbranched_path) { + # merge pairs + $self->_append_node($node, $next_node); + $COMPACTED = 1; + + } + + if ($main::SEE) { + print "Collapsed:\n"; + $self->print_path($node); + print "\n\n\n"; + } + + } + + + $self->validate_graph(); + + + return ($COMPACTED); +} + +#### +sub _append_node { + my $self = shift; + my ($node, $next_node) = @_; + + ## In compacted mode already. + + if (! _compatible_decorations($node, $next_node)) { + confess "Error, trying to join nodes with incompatible decorations."; + } + + my $KmerLength = $self->{KmerLength}; + + + # add counts + $node->set_count($node->get_count() + $next_node->get_count()); + + my $orig_node_seq = $node->get_sequence(); + my $orig_next_node_seq = $next_node->get_sequence(); + + my $new_node_sequence = $orig_node_seq . substr($orig_next_node_seq, $KmerLength-1); # exclude K-1 prefix. + + # add terminal base + $node->set_sequence($new_node_sequence); + + # add read evidence + $node->{_ReadTracker}->append_to_ReadTracker($next_node->{_ReadTracker}); + + # sever connection between node and next + $node->delete_next_node($next_node); + $next_node->delete_prev_node($node); + + # sever connection between next and next-next nodes, and reconnect next-next's with node. + + my @next_node_next_nodes = $next_node->get_all_next_nodes(); + + my %next_next_node_edge_count; + + ## update edge counts. + foreach my $next_next_node (@next_node_next_nodes) { + my $edge_count = $self->get_edge_count($next_node, $next_next_node); + + $next_next_node_edge_count{$next_next_node} = $edge_count; + } + + $self->delete_node_from_graph($next_node); + + foreach my $next_next_node (@next_node_next_nodes) { + $self->link_adjacent_nodes($node, $next_next_node, $next_next_node_edge_count{$next_next_node}); + + } + + + + return; +} + + +#### +sub toGraphViz { + my $self = shift; + + my (%settings) = @_; + + my @nodes = $self->get_all_nodes(); + + my $text = "digraph G {\n"; + + $text .= + #"node [width=0.1,height=0.1,fontsize=10,shape=point];\n" + "node [width=0.1,height=0.1,fontsize=10];\n" + . "edge [fontsize=12];\n" + . "margin=1.0;\n" + . "rankdir=LR;\n" + . "labeljust=l;\n"; + + + foreach my $node (sort {$a->get_ID() cmp $b->get_ID()} @nodes) { + + my @prev_nodes = $node->get_all_prev_nodes(); + my @next_nodes = $node->get_all_next_nodes(); + + my $sequence = $node->get_sequence(); + + if (my $min_len = $settings{no_short_singletons}) { + + if ( scalar(@prev_nodes) == 0 + && + scalar(@next_nodes) == 0 + && + length($sequence) < $min_len) { + + next; + } + } + + + + my $seqLen = length($sequence); + + my $count = $node->get_count(); + + my $len = $seqLen; + + if (scalar(@prev_nodes)==0) { # no left connecting node: + $len = $seqLen - ($self->{KmerLength} - 1); + } + + my $avg_count = $count; #int($count/$len + 0.5); + + my $id = hex($node->get_ID()); + + my $atts = ""; + + if (length($sequence) < $MIN_DISPLAY_SEQ_TAG_LENGTH) { + $atts = "-$sequence"; + } + else { + $atts = "-" . substr($sequence, 0, 3) . "..." . substr($sequence, -3); + } + + my $depth =""; + if (defined $node->{depth}) { + $depth = $node->{depth}; + } + + #$text .= "\t$id \[label=\"$id-L$seqLen-C$avg_count$atts\"$color];\n"; + $text .= "\t$id \[label=\"D:$depth-L$seqLen-C$avg_count$atts\"];\n"; + + + foreach my $next_node (@next_nodes) { + + my $next_id = hex($next_node->get_ID()); + + my @colors_in_common = &get_colors_in_common($node, $next_node); + + my $edge_weight = $self->get_edge_count($node, $next_node); + + if (@colors_in_common) { + foreach my $color (@colors_in_common) { + + my $label = "$edge_weight"; + unless ($color eq 'black') { # reserved for general data + my @accs = @{$self->{color_to_accs}->{$color}}; + $label = join(",", @accs); + } + #$text .= "\t$id->$next_id [label=$edge_weight, color=\"$color\"];\n"; + $text .= "\t$id->$next_id [label=\"$label\", color=\"$color\"];\n"; + + } + } + + } + } + + $text .= "}\n"; + + return($text); +} + + +sub get_colors_in_common { + my ($node, $next_node) = @_; + + my @colorsA = $node->get_colors(); + my @colorsB = $next_node->get_colors(); + + my %counts; + + foreach my $color (@colorsA, @colorsB) { + $counts{$color}++; + } + my @colors = grep { $counts{$_} > 1 } keys %counts; + + return(@colors); +} + + + +#### +sub _prep_graph_for_compaction { + my $self = shift; + + my @nodes = $self->get_all_nodes(); + + my %new_node_set; ## change lookup so it's not based on sequence anymore. + + foreach my $node (@nodes) { + $new_node_set{"$node"} = $node; + } + + $self->{_nodes} = \%new_node_set; +} + + +#### +sub delete_node_from_graph { + my $self = shift; + my ($node) = @_; + + $self->prune_nodes_from_graph($node); + + if ($self->{compacted}) { + + delete $self->{_nodes}->{$node}; + + } + else { + my $kmer = $node->get_sequence(); + delete $self->{_nodes}->{"$kmer"}; + } + + return; +} + + +#### +sub purge_nodes_below_count { + my $self = shift; + + my ($min_count) = @_; + + my @nodes = $self->get_all_nodes(); + + my $num_deleted = 0; + + foreach my $node (@nodes) { + + if ($node->get_count() < $min_count) { + + my @colors = $node->get_colors(); + unless (scalar(@colors) == 1 && $colors[0] eq "black") { + next; + ## skipping base data (reads) below coverage limit + } + + $self->delete_node_from_graph($node); + + $num_deleted++; + } + } + + print STDERR " -purged nodes below count($min_count), deleted $num_deleted nodes.\n"; + + return; +} + + +#### +sub validate_graph { + my $self = shift; + + print STDERR "Validating graph.\n"; + + my @nodes = $self->get_all_nodes(); + + my %node_ids_in_graph; + + foreach my $node (@nodes) { + + my $node_id = $node->get_ID(); + $node_ids_in_graph{$node_id} = $node; + } + + foreach my $node (@nodes) { + + my $id = $node->get_ID(); + + my @prev_nodes = $node->get_all_prev_nodes(); + foreach my $prev_node (@prev_nodes) { + my $prev_id = $prev_node->get_ID(); + + unless (exists ($node_ids_in_graph{$prev_id})) { + print STDERR "Error, $prev_id exists as prev_node to $id, but not in graph\n"; + print STDERR $prev_node->get_sequence() . "\n" . $prev_node->toString(); + confess; + } + + unless ($prev_node->has_next_node($node)) { + confess "Error, prev_node: $prev_id lacks current next node $id as link.\n"; + } + + + } + + my @next_nodes = $node->get_all_next_nodes(); + foreach my $next_node (@next_nodes) { + my $next_id = $next_node->get_ID(); + + unless (exists ($node_ids_in_graph{$next_id})) { + print STDERR "Error, $next_id exists as next_node to $id, but not in graph\n"; + print STDERR $next_node->get_sequence() . "\n" . $next_node->toString(); + confess; + } + + unless ($next_node->has_prev_node($node)) { + confess "Error, next node: $next_id lacks current node $id as prev link.\n"; + } + + + } + + } + + return; +} + +#### +sub print_path { + my $self = shift; + my @nodes = @_; + + my $counter = 0; + my $prev_node; + + foreach my $node (@nodes) { + $counter++; + printf("%4s", $counter); + + my $edge_count = ""; + if ($prev_node) { + $edge_count = " edge_weight: " . $self->get_edge_count($prev_node, $node); + } + + print " " . $node->get_value() . " C:" . $node->get_count() . " " . join(",", $node->get_colors()) . "$edge_count\n"; + + $prev_node = $node; + + } + print "\n"; + + return; +} + + +sub _compatible_decorations { + my ($nodeA, $nodeB) = @_; + + + my @colorsA = $nodeA->get_colors(); + my @colorsB = $nodeB->get_colors(); + + unless (@colorsA || @colorsB) { + # no decorations. + return(1); + } + + my %counts; + foreach my $color (@colorsA, @colorsB) { + $counts{$color}++; + } + + my @colors_unique = grep { $counts{$_} == 1 } keys %counts; + + if (@colors_unique) { + #print STDERR "unique colors: @colors_unique\n"; + return(0); + } + + else { + return(1); + } +} + + + + +#### +sub score_nodes { + my $self = shift; + + $self->assign_depth_to_nodes(); + + ## score = sum prev counts for those reads that are consistent, excluding those that are from bifurcating reads. + + + my @nodes = sort {$a->{depth}<=>$b->{depth}} $self->get_all_nodes(); + + ## assign base scores: + foreach my $node (@nodes) { + + ## define base scores: contributions by reads that are not in prev or next nodes. + my %neighboring_reads = $self->gather_read_indices($node->get_all_prev_nodes(), $node->get_all_next_nodes()); + + my $base_score = 0; + + my $read_tracker = $node->get_ReadTracker(); + my @reads = $read_tracker->get_tracked_read_indices(); + foreach my $read (@reads) { + if ($node->{depth} == 0 || ! $neighboring_reads{$read}) { + my $count = $read_tracker->get_read_base_count($read); + + $base_score += $count; + #print "$node has read [$read] with count [$count]\n"; + + } + } + + $node->{_base_score} = $base_score; + $node->{_sum_score} = $base_score; # init + #print "$node : base score = $base_score\n"; + } + + + ## assign sum scores + foreach my $node (@nodes) { + my %indices_exclude; + + my $read_tracker = $node->get_ReadTracker(); + + my $highest_prev_score = $node->{_sum_score}; + my $highest_prev_node = undef; + + foreach my $prev_node ($node->get_all_prev_nodes()) { + ## want reads that are in current node and prev node, but not other next nodes of prev node (exclusive to current node). + my $sum_score = $prev_node->{_sum_score}; + + my @other_prev_next_nodes = grep { $_ ne $node } $prev_node->get_all_next_nodes(); + + my %reads_ignore = $self->gather_read_indices(@other_prev_next_nodes); + + foreach my $read ($read_tracker->get_tracked_read_indices()) { + + if (! $reads_ignore{$read}) { + + $sum_score += $read_tracker->get_read_base_count($read); + } + } + + push (@{$prev_node->{_forward_scores}}, { next_node => $node, + score => $sum_score, + } ); + + + if ($sum_score > $highest_prev_score) { + $highest_prev_score = $sum_score; + $highest_prev_node = $prev_node; + } + } + + $node->{_sum_score} = $highest_prev_score; + $node->{_best_prev} = $highest_prev_node; + } + + + ## find highest sum score + my $highest_sum_score = 0; + my $best_node = undef; + + foreach my $node (@nodes) { + #print "Node: $node, has sum score: " . $node->{_sum_score} . "\n"; + if ($node->{_sum_score} > $highest_sum_score) { + $highest_sum_score = $node->{_sum_score}; + $best_node = $node; + } + } + + + my @path_nodes; + my $prev_node = $best_node; + while ($prev_node) { + #print "Backtracked: $prev_node " . $prev_node->{_sum_score} . "\n"; + #print $prev_node->toString(); + push (@path_nodes, $prev_node); + $prev_node = $prev_node->{_best_prev}; + + } + + @path_nodes = reverse @path_nodes; + + return (@path_nodes); + + +} + + +#### +sub extract_sequence_from_path { + my $self = shift; + my (@ordered_nodes) = @_; # ordered left to right + + my @seqs; + + foreach my $node (@ordered_nodes) { + + $node->{_visited} = 1; + + push (@seqs, $node->get_sequence()); + + } + + + my $kmer_length = $self->{KmerLength}; + + my $final_seq = shift @seqs; ## first one gets full kmer treatment. + + foreach my $seq (@seqs) { + $seq = substr($seq, $kmer_length-1); + + $final_seq .= $seq; + } + + + return($final_seq); +} + + + +#### +sub gather_read_indices { + my $self = shift; + my (@nodes) = @_; + + my %indices; + + foreach my $node (@nodes) { + + my $read_tracker = $node->get_ReadTracker(); + + my @read_indices = $read_tracker->get_tracked_read_indices(); + + foreach my $read (@read_indices) { + $indices{$read}=1; + } + } + + return(%indices); +} + + +#### +sub assign_depth_to_nodes { + my $self = shift; + print STDERR "Assigning depth to nodes\n"; + + my @root_nodes = $self->get_root_nodes(); + + ## base cases. + foreach my $root (@root_nodes) { + $root->{depth} = 0; + } + + foreach my $node ($self->get_all_nodes()) { + if (! defined ($node->{depth})) { + $self->assign_depth($node, {}); + } + } + + + return; +} + + +#### +sub assign_depth { + my $self = shift; + my $node = shift; + my $seen_href = shift; + + my @ancestors = $node->get_all_prev_nodes(); + + my $max_depth = 0; + + foreach my $ancestor (@ancestors) { + + my $depth; + if (defined($ancestor->{depth})) { + $depth = 1 + $ancestor->{depth}; + } + else { + + if ($seen_href->{$ancestor}) { + # print "Breaking cycle: $ancestor\n" . join("\n", keys %$seen_href) . "\n"; + next; + } + $seen_href->{$ancestor} = 1; + + $depth = 1 + $self->assign_depth($ancestor, $seen_href); + } + if ($depth > $max_depth) { + $max_depth = $depth; + } + } + + # print "$node\tdepth: $max_depth\n"; + $node->{depth} = $max_depth; + return($max_depth); +} + + +#### +sub all_nodes_visited { + my $self = shift; + + my @nodes = $self->get_all_nodes(); + + foreach my $node (@nodes) { + + if (! $node->{_visited}) { + return(0); + } + } + + return(1); # all visited. +} + +#### +sub extract_path_from_unvisited_node { + my $self = shift; + + my @nodes = $self->get_all_nodes(); + + my $highest_scoring_unvisited_node = undef; + my $highest_score = -1; + + foreach my $node (@nodes) { + if (! $node->{_visited}) { + if ($node->{_base_score} > $highest_score) { + $highest_score = $node->{_base_score}; + $highest_scoring_unvisited_node = $node; + } + } + } + + unless ($highest_scoring_unvisited_node) { + confess "Error, no unvisited node found"; + } + + my @best_path; + + my %seen; # don't follow cycles. FIXME: cycles shouldn't be part of the scoring, though... bug somewhere upstream? + + my $prev_node = $highest_scoring_unvisited_node; + while ($prev_node && ! $prev_node->{_visited} && ! $seen{$prev_node}) { + print "walking prev_node: $prev_node\n" if $main::SEE; + $seen{$prev_node} = 1; + unshift(@best_path, $prev_node); + $prev_node = $prev_node->{_best_prev}; + } + + ## now track forward. + my $next_node = $highest_scoring_unvisited_node; + delete $seen{$next_node}; # already used above. + while ($next_node && ! $next_node->{_visited} && ! $seen{$next_node}) { + print "walking next_node: $next_node\n" if $main::SEE; + $seen{$next_node} = 1; + my @structs = @{$next_node->{_forward_scores}}; + if (@structs) { + @structs = sort {$a->{score}<=>$b->{score}} @structs; + my $best_struct = pop @structs; + push (@best_path, $next_node); + $next_node = $best_struct->{next_node}; + } + else { + $next_node = undef; + } + } + + return(@best_path); +} + + + +#### +sub prune_low_weight_edges { + my $self = shift; + my %params = @_; + + my $edge_weight_threshold = $params{edge_weight_threshold}; + + + my @nodes = $self->get_all_nodes(); + + my @edges_to_prune; + + foreach my $node (@nodes) { + my @prev_nodes = $node->get_all_prev_nodes(); + + foreach my $prev_node (@prev_nodes) { + + my $edge_ratio = $self->compute_edge_support_ratio($prev_node, $node); + + #print "EdgeRatio: $edge_ratio\n"; + + if ($edge_ratio < $edge_weight_threshold) { + push (@edges_to_prune, [$prev_node, $node]); + + } + } + } + + + if (@edges_to_prune) { + + foreach my $edges_to_prune (@edges_to_prune) { + + my ($prev_node, $node) = @$edges_to_prune; + $self->prune_edge($prev_node, $node); + } + + return(scalar(@edges_to_prune)); + } + else { + return(0); + } + +} + +#### +sub prune_singletons { + my $self = shift; + + my ($min_seq_length) = @_; + + my @nodes = $self->get_all_nodes(); + + foreach my $node (@nodes) { + + if ( (! $node->get_all_prev_nodes()) + && + (! $node->get_all_next_nodes()) ) { + + if (length ($node->get_sequence()) < $min_seq_length) { + + $self->delete_node_from_graph($node); + } + } + } + + return; +} + + +#### +sub compute_edge_support_ratio { + my $self = shift; + my ($prev_node, $node) = @_; + + ## ratio = reads in common / (all reads out of prev UNION all reads into node) + + my @reads_exiting_prev_node = $prev_node->get_reads_exiting_node(); + my %exiting_reads = map { + $_ => 1 } @reads_exiting_prev_node; + + my @reads_entering_node = $node->get_reads_entering_node(); + my %entering_reads = map { + $_ => 1 } @reads_entering_node; + + my %all_reads = map { + $_ => 1 } (@reads_exiting_prev_node, @reads_entering_node); + + my @all = keys %all_reads; + my @shared; + + foreach my $read (@all) { + if ($entering_reads{$read} && $exiting_reads{$read}) { + push (@shared, $read); + } + } + + my $ratio = scalar(@shared) / scalar(@all); + + return($ratio); +} + + + +#### +sub prune_dangling_nodes { + my $self = shift; + my %params = @_; + + my $min_leaf_node_length = $params{min_leaf_node_length}; + my $min_leaf_node_avg_cov = $params{min_leaf_node_avg_cov}; + + unless (defined ($min_leaf_node_length) + && + defined ($min_leaf_node_avg_cov) ) { + + confess "Error, must define values for both: min_leaf_node_length && min_leaf_node_avg_cov"; + } + + my @nodes = $self->get_all_nodes(); + + my @nodes_to_prune; + + my $kmer_length = $self->{KmerLength}; + + foreach my $node (@nodes) { + + ## check to see if it's a dangling node. + if ( (! $node->get_all_prev_nodes()) + || + (! $node->get_all_next_nodes()) + ) { + + my $length = length($node->get_sequence()); + + my $cov = $node->get_count(); + + my $avg_cov = $cov / ($length - ($kmer_length - 1)); + + if ($avg_cov < $min_leaf_node_avg_cov && $length < $min_leaf_node_length) { + push (@nodes_to_prune, $node); + } + + } + } + + if (@nodes_to_prune) { + + foreach my $node (@nodes_to_prune) { + + $self->prune_nodes_from_graph($node); + } + + return(scalar(@nodes_to_prune)); # nodes pruned. + } + + else { + return(0); # no nodes pruned. + } + +} + + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/KmerNode.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/KmerNode.pm new file mode 100644 index 0000000..2e6f614 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/KmerNode.pm @@ -0,0 +1,173 @@ +package KmerNode; + +use strict; +use warnings; +use Carp; +use ReadTracker; + +use base qw (GenericNode); + +sub new { + my $packagename = shift; + my ($kmer_seq, $accession) = @_; + + my $self = $packagename->SUPER::new($kmer_seq); + + $self->{_count} = 0; + $self->{_ReadTracker} = new ReadTracker(); + + $self->{depth} = undef; + + ## for DP scans: + $self->{_visited} = 0; + $self->{_base_score} = 0; + $self->{_sum_score} = 0; + $self->{_best_prev} = undef; # prev_node in highest scoring path. + $self->{_forward_scores} = []; # stores structs of { node => ref, score => $score} + + + bless ($self, $packagename); + + return($self); +} + + +#### +sub get_ReadTracker { + my $self = shift; + return($self->{_ReadTracker}); +} + + +sub get_sequence { + my $self = shift; + + return($self->get_value()); +} + +sub set_sequence { + my $self = shift; + my $sequence = shift; + + unless ($sequence =~ /\w/) { + confess "Error, need sequence"; + } + + $self->set_value($sequence); + + return; +} + +sub track_reads { + my $self = shift; + my @accs = @_; + + $self->{_ReadTracker}->track_reads(@accs); + + return; +} + +sub get_count { + my $self = shift; + + return($self->{_count}); +} + +sub set_count { + my $self = shift; + + my ($count) = @_; + + $self->{_count} = $count; + return; +} + + + +#### +sub toString { + my $self = shift; + + my @prev_nodes = $self->get_all_prev_nodes(); + + my @next_nodes = $self->get_all_next_nodes(); + + my $text = ""; + + foreach my $prev_node (@prev_nodes) { + $text .= "P " . $prev_node->get_value() . "(" . $prev_node->get_count() . ") $prev_node " . join(",", $prev_node->get_colors()) . "\n"; + } + $text .= "X " . $self->get_value() . "(" . $self->get_count() . ") $self " . join(",", $self->get_colors()) . "\n"; + + foreach my $next_node (@next_nodes) { + $text .= "N " . $next_node->get_value() . "(" . $self->get_count() . ") $next_node " . join(",", $next_node->get_colors()) . "\n"; + } + + return($text); +} + + +#### +sub get_reads_exiting_node { + my $self = shift; + + my @next_nodes = $self->get_all_next_nodes(); + unless (@next_nodes) { + return(); + } + + my $read_tracker = $self->get_ReadTracker(); + + my @reads_in_node = $read_tracker->get_tracked_read_indices(); + + my @reads_in_next_nodes; + foreach my $next_node (@next_nodes) { + push (@reads_in_next_nodes, $next_node->get_ReadTracker()->get_tracked_read_indices()); + } + + my %next_node_reads = map { + $_ => 1 } @reads_in_next_nodes; + + my @exiting_reads; + foreach my $read (@reads_in_node) { + if ($next_node_reads{$read}) { + push (@exiting_reads, $read); + } + } + + return(@exiting_reads); +} + +#### +sub get_reads_entering_node { + my $self = shift; + + my @prev_nodes = $self->get_all_prev_nodes(); + unless (@prev_nodes) { + return(); + } + + my $read_tracker = $self->get_ReadTracker(); + + my @reads_in_node = $read_tracker->get_tracked_read_indices(); + + my @reads_in_prev_nodes; + foreach my $prev_node (@prev_nodes) { + push (@reads_in_prev_nodes, $prev_node->get_ReadTracker()->get_tracked_read_indices()); + } + + my %prev_node_reads = map { + $_ => 1 } @reads_in_prev_nodes; + + my @entering_reads; + foreach my $read (@reads_in_node) { + if ($prev_node_reads{$read}) { + push (@entering_reads, $read); + } + } + + return(@entering_reads); +} + + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/ReadCoverageGraph.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/ReadCoverageGraph.pm new file mode 100644 index 0000000..0dad284 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/ReadCoverageGraph.pm @@ -0,0 +1,244 @@ +package main; +our $SEE; + +package ReadCoverageGraph; + +use strict; +use warnings; +use Carp; + +use base qw(GenericGraph); +use ReadCoverageNode; + +no warnings qw (recursion); + +sub new { + my $packagename = shift; + + my $self = $packagename->SUPER::new(); + + bless ($self, $packagename); + + return($self); +} + + + +sub get_or_create_node { + my $self = shift; + my ($node_name, $read_accession) = @_; + + if ($self->node_exists($node_name)) { + my $node = $self->get_node($node_name); + $node->add_read($read_accession); + return($node); + + } + else { + # instantiate it, add it to the graph + return($self->create_node($node_name, $read_accession)); + } +} + + +sub create_node { + my $self = shift; + + my ($node_name, $read_accession) = @_; + + if ($self->node_exists($node_name)) { + confess "Error, node $node_name already exists in the graph"; + } + + my $node = new ReadCoverageNode($node_name, $read_accession); + + $self->{_nodes}->{$node_name} = $node; + + return($node); +} + + +#### +sub get_nodes_sorted_by_count_desc { + my $self = shift; + my @nodes = $self->get_all_nodes(); + + @nodes = reverse sort {$a->{_count}<=>$b->{_count}} @nodes; + + return(@nodes); +} + + +sub find_maximal_path_including_node { + my $self = shift; + + my ($nucleating_node, $max_recurse_depth) = @_; + + my $path_forward_aref = [$nucleating_node]; + my $depth_forward = 0; + my $sum_forward_count = 0; + do { + my $start_node = $path_forward_aref->[-1]; + ($path_forward_aref, $sum_forward_count, $depth_forward) = $self->extend_path("next", $start_node, $path_forward_aref, $max_recurse_depth, 0, 0); + + if ($main::SEE) { + print "Forward.\n"; + &print_path(@$path_forward_aref); + } + + } while ($depth_forward > 0); + + if ($main::SEE) { + print "Forward, done.\n"; + &print_path(@$path_forward_aref); + } + + + my $path_reverse_aref = [$nucleating_node]; + my $depth_reverse = 0; + my $sum_reverse_count = 0; + do { + my $start_node = $path_reverse_aref->[-1]; + ($path_reverse_aref, $sum_reverse_count, $depth_reverse) = $self->extend_path("prev", $start_node, $path_reverse_aref, $max_recurse_depth, 0, 0); + + if ($main::SEE) { + print "Reverse:\n"; + &print_path(@$path_reverse_aref); + } + + } while ($depth_reverse > 0); + + + if ($main::SEE) { + print "Reverse, done.\n"; + &print_path(@$path_reverse_aref); + } + + ## unwrap path + + # pull out the nucleating node, should be first one in each path. + shift @$path_forward_aref; + shift @$path_reverse_aref; + + my @path = ( (reverse @$path_reverse_aref), $nucleating_node, @$path_forward_aref); + + if ($main::SEE) { + print "Done.\n"; + &print_path(@path); + } + + return(@path); + +} + + +#### +sub extend_path { + my $self = shift; + + my ($direction, + $node, + $curr_path_list_aref, + $max_recurse_depth, + $sum_counts, + $curr_depth) = @_; + + + my $path_length = scalar (@$curr_path_list_aref); + + print "extending $direction from " . $node->get_value() . ", K:$path_length S:$sum_counts, D:$curr_depth\n" if $main::SEE; + + #print join("\t", @_) . "\n"; + + ## curr_path_list_aref should include the incoming node already + + if ($curr_depth >= $max_recurse_depth) { + return($curr_path_list_aref, $sum_counts, $curr_depth); + } + + ## explore connected nodes + my @other_nodes; + if ($direction eq 'next') { + @other_nodes = $node->get_all_next_nodes(); + } + else { + @other_nodes = $node->get_all_prev_nodes(); + } + + ## only explore those other_node's that are not already seen along the current path + my @unseen_nodes; + + foreach my $other_node (@other_nodes) { + unless (grep {$_ == $other_node} @$curr_path_list_aref) { + push (@unseen_nodes, $other_node); + } + } + + #print "Curr path nodes: " . join(" ", @$curr_path_list_aref) . "\n"; + + @other_nodes = @unseen_nodes; # reset to those that haven't been seen already. + + #print "\tOther nodes: " . join(" ", @other_nodes) . "\n\n"; + + if (@other_nodes) { + ## examine possible paths: + + if ($main::SEE) { + print "Extending from:\n" . $node->get_value() . " (" . $node->get_count() . ") Depth:$curr_depth to\n"; + foreach my $other_node (@other_nodes) { + print $other_node->get_value() . " (" . $other_node->get_count() . ")\n"; + } + print "\n"; + } + + + my @alt_paths; + foreach my $other_node (@other_nodes) { + #print $other_node->toString() . "\n"; + + my ($path_list_aref, $counts, $depth) = $self->extend_path($direction, + $other_node, + [@$curr_path_list_aref, $other_node], # tack it on to the list + $max_recurse_depth, + $sum_counts + $other_node->get_count(), + $curr_depth+1); + + push (@alt_paths, [$path_list_aref, $counts, $depth]); + } + + ## select the greatest one + @alt_paths = sort { #$a->[2] <=> $b->[2] ## Perhaps include extension length + # || + $a->[1] <=>$b->[1] } @alt_paths; + + my $top_path = pop @alt_paths; + + my ($path_list_aref, $counts, $depth) = @$top_path; + return($path_list_aref, $counts, $depth); + } + else { + ## no extensions possible + return($curr_path_list_aref, $sum_counts, $curr_depth); + } +} + + + +#### +sub print_path { + my @nodes = @_; + + my $counter = 0; + foreach my $node (@nodes) { + $counter++; + printf("%4s", $counter); + print " " . $node->get_value() . " C:" . $node->get_count() . "\n"; + } + print "\n"; + + return; +} + + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/ReadCoverageNode.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/ReadCoverageNode.pm new file mode 100644 index 0000000..541378b --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/ReadCoverageNode.pm @@ -0,0 +1,188 @@ +package ReadCoverageNode; + +use strict; +use warnings; + +use Carp; +use base qw (GenericNode); + + +## static vars +my %READ_TRACKER; +my $READ_COUNTER = 0; + + +sub new { + my $packagename = shift; + my ($node_name, $read_acc) = @_; + + my $self = $packagename->SUPER::new($node_name); # sets _value + + + + $self->{_reads} = {}; + $self->{_count} = 0; + + + bless ($self, $packagename); + + $self->add_read($read_acc); + + + return($self); +} + +#### +sub get_count { + my $self = shift; + return($self->{_count}); +} + + +sub set_count { + my $self = shift; + my $count = shift; + + $self->{_count} = $count; + return; +} + + +#### +sub add_read { + my $self = shift; + my $read_acc = shift; + + + my $node_name = $self->get_value(); + unless ((defined $read_acc) && $read_acc =~ /\w/) { + confess "Error, need read accession"; + } + + my $read_tracking_no = $self->_get_or_create_tracking_number($read_acc); + + if (exists $self->{_reads}->{$read_tracking_no}) { + + #print "Already got read $read_acc for $node_name, count:" . $self->get_count() . "\n"; + + } + else { + $self->{_reads}->{$read_tracking_no} = 1; + $self->{_count}++; + + #print "-incrementing count for $read_acc for $node_name => $self->{_count}\n"; + + } + + return; +} + +#### +sub get_reads { + my $self = shift; + + my @reads = keys %{$self->{_reads}}; + + return(@reads); +} + +### +sub has_read { + my $self = shift; + my $read_acc = shift; + + if (exists $self->{_reads}->{$read_acc}) { + return(1); + } + else { + return(0); + } +} + + + +#### +sub count_reads_in_common { + my $self = shift; + my $other_node = shift; + + my $count = 0; + + foreach my $read ($self->get_reads()) { + if ($other_node->has_read($read)) { + $count++; + } + } + + return($count); +} + +#### +sub toString { + my $self = shift; + + my @prev_nodes = $self->get_all_prev_nodes(); + + my @next_nodes = $self->get_all_next_nodes(); + + my $text = ""; + + foreach my $prev_node (@prev_nodes) { + $text .= "P " . $prev_node->get_value() . "(" . $prev_node->get_count() . ")\n"; + } + $text .= "X " . $self->get_value() . "(" . $self->get_count() . ")\n"; + + foreach my $next_node (@next_nodes) { + $text .= "N " . $next_node->get_value() . "(" . $self->get_count() . ")\n"; + } + + return($text); +} + +#### Private read tracking ## this should probably be a separate class at some point, with a singleton class object. + +sub _read_is_tracked { + my $self = shift; + my $read_acc = shift; + + if (exists $READ_TRACKER{$read_acc}) { + return(1); + } + else { + return(0); + } +} + +sub _get_read_tracking_number { + my $self = shift; + my $read_acc = shift; + + if (! $self->_read_is_tracked($read_acc)) { + die "Error, read $read_acc is not tracked"; + } + + my $tracking_number = $READ_TRACKER{$read_acc}; + return($tracking_number); +} + +sub _get_or_create_tracking_number { + my $self = shift; + my $read_acc = shift; + + if ($self->_read_is_tracked($read_acc)) { + return($self->_get_read_tracking_number($read_acc)); + } + else { + ## track this new read: + $READ_COUNTER++; + $READ_TRACKER{$read_acc} = $READ_COUNTER; + return($self->_get_read_tracking_number($read_acc)); + } +} + + + + + + +1; # EOM diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/ReadManager.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/ReadManager.pm new file mode 100644 index 0000000..bd02426 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/ReadManager.pm @@ -0,0 +1,31 @@ +package ReadManager; + +use strict; +use warnings; +use Carp; + + +my $counter = 0; +my %read_acc_to_counter; + + +#### +sub get_read_index { + my ($read_acc) = @_; + + if (my $read_index = $read_acc_to_counter{$read_acc}) { + + return($read_index); + } + + else { + + $counter++; + $read_acc_to_counter{$read_acc} = $counter; + + return($counter); + } +} + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/ReadTracker.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/ReadTracker.pm new file mode 100644 index 0000000..a10d457 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/ReadTracker.pm @@ -0,0 +1,67 @@ +package ReadTracker; + +use strict; +use warnings; +use Carp; +use ReadManager; + + +sub new { + my ($packagename) = shift; + + + my $self = { reads => {}, # read indices, use ReadManager to hold full acc strings. + + }; + + bless ($self, $packagename); + + return($self); +} + +sub track_reads { + my $self = shift; + my @reads = @_; + + foreach my $read (@reads) { + my $index = &ReadManager::get_read_index($read); + $self->{reads}->{$index}++; + } + + return; +} + +sub append_to_ReadTracker { + my $self = shift; + my $to_add_ReadTracker = shift; + + foreach my $read_index (keys %{$to_add_ReadTracker->{reads}}) { + $self->{reads}->{$read_index} += $to_add_ReadTracker->{reads}->{$read_index}; + } + + return; +} + +sub get_tracked_read_indices { + my $self = shift; + + return(keys %{$self->{reads}}); +} + +sub get_read_base_count { + my $self = shift; + my $read = shift; + + unless (defined $self->{reads}->{$read}) { + confess "Error, no read index stored [$read] "; + } + + return($self->{reads}->{$read}); +} + + + +1; #EOM + + + diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/SAM_entry.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/SAM_entry.pm new file mode 100644 index 0000000..225fa0d --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/SAM_entry.pm @@ -0,0 +1,411 @@ +package SAM_entry; + +use strict; +use warnings; +use Carp; + + + +sub new { + my $packagename = shift; + my ($line) = @_; + + unless (defined $line) { + confess "Error, need sam text line as parameter"; + } + + chomp $line; + + my @fields = split(/\t/, $line); + + my $self = { + _fields => [@fields], + }; + + bless ($self, $packagename); + + return($self); +} + + +#### +sub get_read_name { + my $self = shift; + return ($self->{_fields}->[0]); +} + + +#### +sub get_scaffold_name { + my $self = shift; + return($self->{_fields}->[2]); +} + +#### +sub get_aligned_position { + my $self = shift; + return($self->{_fields}->[3]); +} + +sub get_scaffold_position { # preferred + my $self = shift; + return($self->get_aligned_position()); +} + + +#### +sub get_cigar_alignment { + my $self = shift; + return($self->{_fields}->[5]); +} + +#### +sub get_alignment_coords { + my $self = shift; + + my $genome_lend = $self->get_aligned_position(); + + my $alignment = $self->get_cigar_alignment(); + + my $query_lend = 0; + + my @genome_coords; + my @query_coords; + + + $genome_lend--; # move pointer just before first position. + + while ($alignment =~ /(\d+)([A-Z])/g) { + my $len = $1; + my $code = $2; + + unless ($code =~ /^[MSDNIH]$/) { + confess "Error, cannot parse cigar code [$code]"; + } + + # print "parsed $len,$code\n"; + + if ($code eq 'M' || $code eq 'S' || $code eq 'H') { # aligned bases match or mismatch + + my $genome_rend = $genome_lend + $len; + my $query_rend = $query_lend + $len; + + push (@genome_coords, [$genome_lend+1, $genome_rend]); + push (@query_coords, [$query_lend+1, $query_rend]); + + # reset coord pointers + $genome_lend = $genome_rend; + $query_lend = $query_rend; + + } + elsif ($code eq 'D' || $code eq 'N') { # insertion in the genome, gap in query (intron, perhaps) + $genome_lend += $len; + + } + + elsif ($code eq 'I') { # gap in genome, insertion in query + + $query_lend += $len; + + } + } + + return(\@genome_coords, \@query_coords); +} + + +#### +sub get_mate_scaffold_name { + my $self = shift; + + return($self->{_fields}->[6]); +} + + +#### +sub set_mate_scaffold_name { + my $self = shift; + my $mate_scaffold_name = shift; + + $self->{_fields}->[6] = $mate_scaffold_name; + + return; +} + + +#### +sub get_mate_scaffold_position { + my $self = shift; + + return($self->{_fields}->[7]); +} + + +#### +sub set_mate_scaffold_position { + my $self = shift; + my $scaff_pos = shift; + + $self->{_fields}->[7] = $scaff_pos; + + return; +} + + +#### +sub toString { + my $self = shift; + return( join("\t", @{$self->{_fields}}) ); +} + + +#### +sub get_mapping_quality { + my $self = shift; + return($self->{_fields}->[4]); +} + + +#### +sub get_sequence { + my $self = shift; + return($self->{_fields}->[9]); +} + +#### +sub get_quality_scores { + my $self = shift; + return($self->{_fields}->[10]); +} + + + +################### +## Flag Processing +################### + +# from sam format spec: + +=flag_description + +Flag Description +0x0001 the read is paired in sequencing, no matter whether it is mapped in a pair +0x0002 the read is mapped in a proper pair (depends on the protocol, normally inferred during alignment) 1 +0x0004 the query sequence itself is unmapped +0x0008 the mate is unmapped 1 +0x0010 strand of the query (0 for forward; 1 for reverse strand) +0x0020 strand of the mate 1 +0x0040 the read is the first read in a pair 1,2 +0x0080 the read is the second read in a pair 1,2 +0x0100 the alignment is not primary (a read having split hits may have multiple primary alignment records) +0x0200 the read fails platform/vendor quality checks +0x0400 the read is either a PCR duplicate or an optical duplicate + +1. Flag 0x02, 0x08, 0x20, 0x40 and 0x80 are only meaningful when flag 0x01 is present. +2. If in a read pair the information on which read is the first in the pair is lost in the upstream analysis, flag 0x01 shuld +be present and 0x40 and 0x80 are both zero. + +=cut + + +#### +sub get_flag { + my $self = shift; + my $flag = $self->{_fields}->[1]; + return($flag); +} + +sub set_flag { + my $self = shift; + my $flag = shift; + + unless (defined $flag) { + confess "Error, need flag value"; + } + + $self->{_fields}->[1] = $flag; + return; +} + +#### +sub is_paired { + my $self = shift; + return($self->_get_bit_val(0x0001)); +} + +sub set_paired { + my $self = shift; + my $bit_val = shift; + + $self->_set_bit_val(0x0001, $bit_val); + + return; +} + +#### +sub is_proper_pair { + my $self = shift; + return($self->_get_bit_val(0x0002)); +} + +sub set_proper_pair { + my $self = shift; + my $bit_val = shift; + + $self->_set_bit_val(0x0002, $bit_val); + return; +} + +#### +sub is_query_unmapped { + my $self = shift; + return($self->_get_bit_val(0x0004)); +} + +sub set_query_unmapped { + my $self = shift; + my $bit_val = shift; + + $self->_set_bit_val(0x0004, $bit_val); +} + + +#### +sub is_mate_unmapped { + my $self = shift; + return($self->_get_bit_val(0x0008)); +} + +sub set_mate_unmapped { + my $self = shift; + my $bit_val = shift; + + return($self->_set_bit_val(0x0008, $bit_val)); +} + +#### +sub get_query_strand { + my $self = shift; + + my $strand = ($self->_get_bit_val(0x0010)) ? '-' : '+'; + return($strand); +} + +#### +sub get_query_transcribed_strand { + my $self = shift; + + my $strand = $self->get_query_strand(); + + if ($self->is_paired() && $self->is_first_in_pair()) { + + my $transcribed_strand = ($strand eq '+') ? '-' : '+'; + + return($transcribed_strand); + } + else { + return($strand); + } +} + + + + +sub set_query_strand { + my $self = shift; + my $strand = shift; + + unless ($strand eq '+' || $strand eq '-') { + confess "Error, strand value must be [+-]"; + } + + my $bit_val = ($strand eq '+') ? 0 : 1; + $self->_set_bit_val(0x0010, $bit_val); +} + +#### +sub get_mate_strand { + my $self = shift; + + my $strand = ($self->_get_bit_val(0x0020)) ? '-' : '+'; + return($strand); +} + +sub set_mate_strand { + my $self = shift; + my $strand = shift; + + unless ($strand eq '+' || $strand eq '-') { + confess "Error, strand value must be [+-]"; + } + + my $bit_val = ($strand eq '+') ? 0 : 1; + $self->_set_bit_val(0x0020, $bit_val); +} + +#### +sub is_first_in_pair { + my $self = shift; + return($self->_get_bit_val(0x0040)); +} + +sub set_first_in_pair { + my $self = shift; + my $bit_val = shift; + + $self->_set_bit_val(0x0040, $bit_val); + return; +} + +#### +sub is_second_in_pair { + my $self = shift; + return($self->_get_bit_val(0x0080)); +} + + +sub set_second_in_pair { + my $self = shift; + my $bit_val = shift; + + $self->_set_bit_val(0x0080, $bit_val); + return; +} + + + +#### +sub _get_bit_val { + my $self = shift; + my ($bit_position) = @_; + + my $flag = $self->get_flag(); + return($flag & $bit_position); +} + + +#### +sub _set_bit_val { + my $self = shift; + my ($bit_position, $bit_val) = @_; + + unless (defined $bit_position && defined $bit_val) { + confess "Error, need bit position and value"; + } + + my $flag = $self->get_flag(); + + if ($bit_val) { + $flag |= $bit_position; + } + else { + # erase bit + $flag &= ~$bit_position; + } + + $self->set_flag($flag); +} + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/SAM_reader.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/SAM_reader.pm new file mode 100644 index 0000000..9c0d8cf --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/SAM_reader.pm @@ -0,0 +1,98 @@ +package SAM_reader; + +use strict; +use warnings; +use Carp; + +use SAM_entry; + +sub new { + my $packagename = shift; + my $filename = shift; + + unless ($filename) { + confess "Error, need SAM filename as parameter"; + } + + my $self = { filename => $filename, + _next => undef, + _fh => undef, + }; + + bless ($self, $packagename); + + $self->_init(); + + return($self); +} + + +#### +sub _init { + my ($self) = @_; + + open ($self->{_fh}, $self->{filename}) or confess "Error, cannot open file " . $self->{filename}; + + $self->_advance(); + + return; +} + +#### +sub _advance { + my ($self) = @_; + + my $fh = $self->{_fh}; + + my $next_line = <$fh>; + + if ($next_line) { + $self->{_next} = new SAM_entry($next_line); + } + else { + $self->{_next} = undef; + } + + return; +} + +#### +sub has_next { + my $self = shift; + + if (defined $self->{_next}) { + return(1); + } + else { + return(0); + } +} + + +#### +sub get_next { + my $self = shift; + + my $next_entry = $self->{_next}; + + $self->_advance(); + + if (defined $next_entry) { + return($next_entry); + } + else { + return(undef); + } +} + +#### +sub preview_next { + my $self = shift; + return($self->{_next}); +} + + +1; + + + diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/SAM_to_AlignGraph.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/SAM_to_AlignGraph.pm new file mode 100644 index 0000000..e933a01 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/SAM_to_AlignGraph.pm @@ -0,0 +1,50 @@ +package SAM_to_AlignGraph; + +## Static class + +use strict; +use warnings; + +use AlignGraph; +use Carp; + +use SAM_reader; +use SAM_entry; + +sub construct_AlignGraph { + my ($sam_file) = @_; + + my $graph = new AlignGraph(); + + my $sam_reader = new SAM_reader($sam_file); + + my $counter = 0; + + while ($sam_reader->has_next()) { + + $counter++; + print STDERR "\r[$counter] " if $counter % 100 == 0; + + my $sam_entry = $sam_reader->get_next(); + + my $scaff = $sam_entry->get_scaffold_name(); + my $read_acc = $sam_entry->get_read_name(); + + if ($sam_entry->is_query_unmapped()) { next; } + + + + my $query_strand = $sam_entry->get_query_transcribed_strand(); + + my ($genome_coords_aref, $query_coords_aref) = $sam_entry->get_alignment_coords(); + + $graph->add_alignment($read_acc, $scaff, $query_strand, $genome_coords_aref); + + } + + return($graph); +} + +1; #EOM + + diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/StringGraph.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/StringGraph.pm new file mode 100644 index 0000000..c55a2a5 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/StringGraph.pm @@ -0,0 +1,1155 @@ +package main; +our $SEE; + +package StringGraph; + +use strict; +use warnings; +use Carp; +use StringNode; +use ReadTracker; + +use base qw (ReadCoverageGraph); + +no warnings qw (recursion); + + +my $MIN_DISPLAY_SEQ_TAG_LENGTH = 10; # lower than this, and shows up in the dot file as a label. + +sub new { + my $packagename = shift; + + my $self = { + _nodes => {}, #kmerSeq => nodeAddress until after compaction... becomes nodeaddress => $nodeaddress + compacted => 0, + + _edge_color_labels => {}, # keyed on (prev_node, next_node, color) + + _indirect_edge_color_labels => {}, # keyed on color only + + }; + + bless ($self, $packagename); + + return($self); +} + + +sub add_sequence_to_graph { + my $self = shift; + my ($acc, $sequence_node_names_aref, $weight, $color) = @_; ## Sequence type can be undef, ReferenceTranscript, or GreedyPath + + unless (defined($weight) && $weight =~ /^\d+/) { + confess "Error, params: (seq, weight, color), and weight must be numeric."; + } + + if ($self->{compacted}) { + confess "Cannot add more sequences to the graph after it has been compacted"; + } + + + my $prev_node_name = $sequence_node_names_aref->[0]; + my $prev_node = $self->get_or_create_node($prev_node_name); + $prev_node->{_count}++; + $prev_node->track_reads($acc); + + if ($color) { + $prev_node->add_color($color); + } + + for (my $i = 1; $i <= $#$sequence_node_names_aref; $i++) { + my $next_node_name = $sequence_node_names_aref->[$i]; + my $next_node = $self->get_or_create_node($next_node_name); + $next_node->track_reads($acc); + + if ($color) { + $next_node->add_color($color); + } + + $next_node->{_count}++; + + my $prev_edge_weight = $self->get_edge_count($prev_node, $next_node); + + $self->_set_edge_color_label($prev_node, $next_node, $color, $acc); + + ## only link together kmers that contain only recognizable nucleotides + $self->link_adjacent_nodes($prev_node, $next_node, $weight); + + my $edge_weight = $self->get_edge_count($prev_node, $next_node); + if ($edge_weight < $weight) { + confess "Error, added weight $weight to edge $prev_node -> $next_node and didn't stick"; + } + + if ($prev_edge_weight > $edge_weight) { + confess "Error, after adding new edge, prev weight of $prev_edge_weight became: $edge_weight "; + } + + + $prev_node = $next_node; + $prev_node_name = $next_node_name; + } + + return; +} + +sub get_or_create_node { + my $self = shift; + my ($node_name, $read_accession) = @_; + + if ($self->node_exists($node_name)) { + my $node = $self->get_node($node_name); + return($node); + + } + else { + # instantiate it, add it to the graph + return($self->create_node($node_name)); + } +} + + +sub create_node { + my $self = shift; + + my ($node_name) = @_; + + if ($self->node_exists($node_name)) { + confess "Error, node $node_name already exists in the graph"; + } + + my $node = new StringNode($node_name); + + $self->{_nodes}->{$node_name} = $node; + + return($node); +} + + + + +#### +sub compact_graph { + my $self = shift; + + $self->validate_graph(); + + my $kmer_length = $self->{KmerLength}; + + unless ($self->{compacted}) { + $self->_prep_graph_for_compaction(); + + $self->{compacted} = 1; + } + + my $COMPACTED = 0; # flag + + + ## identify all nodes that are branched on the left and not branched on the right + ## then join together the unbranched nodes starting from these and walking right. + + my @nodes = $self->get_all_nodes(); + + ## merge unbranched nodes in runs. + + + + + #### + #### \ + #### (-)--- + #### / + + # or + + # (-)--- + + + # or + + #### (-)------------ + #### / + #### - + #### \ + #### (-) ------------ + + + # or, single line with incompatible decorations: + + # aaaaaaaaaa b bbbbbbbbbb + # ----------(-)---------- + + + my @init_nodes; + foreach my $node (@nodes) { + my @prev_nodes = $node->get_all_prev_nodes(); + + my @next_nodes = $node->get_all_next_nodes(); + + if + + ( (scalar(@next_nodes) == 1) + && + + ( + + + + + ( + (scalar(@prev_nodes) != 1) # branched before or no node before. + || + (scalar (@prev_nodes) == 1 && scalar($prev_nodes[0]->get_all_next_nodes() > 1) ) # previous node branches off. + + ) + + || + + ## decoration switch + (scalar(@prev_nodes) == 1 && (! &_compatible_decorations($node, $prev_nodes[0])) && (&_compatible_decorations($node, $next_nodes[0])) ) + + ) + + ) + { + # got one + push (@init_nodes, $node); + } + + } + + print "Got " . scalar(@init_nodes) . " init nodes.\n" if $main::SEE; + + #foreach my $node (@init_nodes) { + # print "Init node $node:\n" . $node->toString() . "\n";# if $main::SEE; + #} + + + ## Collapse neighboring nodes before branching + foreach my $node (@init_nodes) { + + #print "Init node $node:\n" . $node->toString() . "\n";# if $main::SEE; + + my @unbranched_path; + + my ($next_node) = $node->get_all_next_nodes(); + + unless ($next_node) { + print "warning, init node now lacks next nodes... skipping.\n"; # potential bug? or circular dealt with already? + next; + } + + my %seen = ($node => 1); + + while (1) { + # avoid circularity + if ($seen{$next_node}) { + print "Seen $next_node already... avoiding circularity.\n" if $main::SEE; + last; + } + $seen{$next_node} = 1; + + unless (&_compatible_decorations($node, $next_node)) { + print "Incompatible decorations.\n" if $main::SEE; + last; + } + + # check for left branching on next node + my $num_prev_nodes = scalar ($next_node->get_all_prev_nodes()); + if (scalar $num_prev_nodes > 1) { + print "Got $num_prev_nodes prev nodes for $next_node; terminating extension.\n" if $main::SEE; + last; + } + + push (@unbranched_path, $next_node); + my @next_nodes = $next_node->get_all_next_nodes(); + + if (scalar @next_nodes != 1) { + last; + } + $next_node = shift @next_nodes; + + } + + if ($main::SEE) { + print "Single unbranched path:\n"; + $self->print_path($node, @unbranched_path); + } + + foreach my $next_node (@unbranched_path) { + # merge pairs + $self->_append_node($node, $next_node); + $COMPACTED = 1; + + } + + if ($main::SEE) { + print "Collapsed:\n"; + $self->print_path($node); + print "\n\n\n"; + } + + } + + + $self->validate_graph(); + + + return ($COMPACTED); +} + +#### +sub _append_node { + my $self = shift; + my ($node, $next_node) = @_; + + ## In compacted mode already. + + if (! _compatible_decorations($node, $next_node)) { + confess "Error, trying to join nodes with incompatible decorations."; + } + + my $KmerLength = $self->{KmerLength}; + + + # add counts + $node->set_count($node->get_count() + $next_node->get_count()); + + my $orig_node_seq = $node->get_sequence(); + my $orig_next_node_seq = $next_node->get_sequence(); + + my $new_node_sequence = $orig_node_seq . substr($orig_next_node_seq, $KmerLength-1); # exclude K-1 prefix. + + # add terminal base + $node->set_sequence($new_node_sequence); + + # add read evidence + $node->{_ReadTracker}->append_to_ReadTracker($next_node->{_ReadTracker}); + + # sever connection between node and next + $node->delete_next_node($next_node); + $next_node->delete_prev_node($node); + + # sever connection between next and next-next nodes, and reconnect next-next's with node. + + my @next_node_next_nodes = $next_node->get_all_next_nodes(); + + my %next_next_node_edge_count; + + ## update edge counts. + foreach my $next_next_node (@next_node_next_nodes) { + my $edge_count = $self->get_edge_count($next_node, $next_next_node); + + $next_next_node_edge_count{$next_next_node} = $edge_count; + } + + $self->delete_node_from_graph($next_node); + + foreach my $next_next_node (@next_node_next_nodes) { + $self->link_adjacent_nodes($node, $next_next_node, $next_next_node_edge_count{$next_next_node}); + + } + + + + return; +} + + +sub _get_edge_color_label_key { + my $self = shift; + my ($prev_node, $next_node, $color) = @_; + + my $edge_color_label = join("$;", $prev_node, $next_node, $color); + + return($edge_color_label); + +} + +sub _set_edge_color_label { + my $self = shift; + my ($prev_node, $next_node, $color, $label) = @_; + + my $key = $self->_get_edge_color_label_key($prev_node, $next_node, $color); + + $self->{_edge_color_labels}->{$key} = $label; + + $self->{_indirect_edge_color_labels}->{$color} = $label; + + return; +} + +sub _get_edge_color_label { + my $self = shift; + + my ($prev_node, $next_node, $color) = @_; + + my $key = $self->_get_edge_color_label_key($prev_node, $next_node, $color); + + my $label = $self->{_edge_color_labels}->{$key}; + + return($label); + +} + +sub _get_indirect_edge_color_label { + my $self = shift; + my $color = shift; + + return($self->{_indirect_edge_color_labels}->{$color}); + +} + + +#### +sub toGraphViz { + my $self = shift; + + my (%settings) = @_; + + my @nodes = $self->get_all_nodes(); + + my $text = "digraph G {\n"; + + $text .= + #"node [width=0.1,height=0.1,fontsize=10,shape=point];\n" + "node [width=0.1,height=0.1,fontsize=10];\n" + . "edge [fontsize=12];\n" + . "margin=1.0;\n" + . "rankdir=LR;\n" + . "labeljust=l;\n"; + + + foreach my $node (sort {$a->get_ID() cmp $b->get_ID()} @nodes) { + + my @prev_nodes = $node->get_all_prev_nodes(); + my @next_nodes = $node->get_all_next_nodes(); + + my $sequence = $node->get_sequence(); + + #if (my $min_len = $settings{no_short_singletons}) { + # + # if ( scalar(@prev_nodes) == 0 + # && + # scalar(@next_nodes) == 0 + # && + # length($sequence) < $min_len) { + # + # next; + # } + #} + + my $seqLen = length($sequence); + + my $count = $node->get_count(); + + my $avg_count = $count; #int($count/$len + 0.5); + + my $id = hex($node->get_ID()); + + my $depth =""; + if (defined $node->{depth}) { + $depth = $node->{depth}; + } + + #$text .= "\t$id \[label=\"$id-L$seqLen-C$avg_count$atts\"$color];\n"; + $text .= "\t$id \[label=\"$sequence:D$depth-C$avg_count\"];\n"; + + + foreach my $next_node (@next_nodes) { + + my $next_id = hex($next_node->get_ID()); + + my @colors_in_common = &get_colors_in_common($node, $next_node); + + #my $edge_weight = $self->get_edge_count($node, $next_node); + + + if (@colors_in_common) { + foreach my $color (@colors_in_common) { + my $edge_label = $self->_get_edge_color_label($node, $next_node, $color); + + if ($edge_label) { + $text .= "\t$id->$next_id [label=\"$edge_label\", color=\"$color\"];\n"; + } + else { + ### look for indirect edge + #my $indirect_edge_label = $self->_get_indirect_edge_color_label($color); + #$text .= "\t$id->$next_id [label=\"$indirect_edge_label\", color=\"$color\", style=\"dashed\"];\n"; + } + } + } + + } + } + + $text .= "}\n"; + + return($text); +} + + +sub get_colors_in_common { + my ($node, $next_node) = @_; + + my @colorsA = $node->get_colors(); + my @colorsB = $next_node->get_colors(); + + my %counts; + + foreach my $color (@colorsA, @colorsB) { + $counts{$color}++; + } + my @colors = grep { $counts{$_} > 1 } keys %counts; + + return(@colors); +} + + + +#### +sub _prep_graph_for_compaction { + my $self = shift; + + my @nodes = $self->get_all_nodes(); + + my %new_node_set; ## change lookup so it's not based on sequence anymore. + + foreach my $node (@nodes) { + $new_node_set{"$node"} = $node; + } + + $self->{_nodes} = \%new_node_set; +} + + +#### +sub delete_node_from_graph { + my $self = shift; + my ($node) = @_; + + $self->prune_nodes_from_graph($node); + + if ($self->{compacted}) { + + delete $self->{_nodes}->{$node}; + + } + else { + my $kmer = $node->get_sequence(); + delete $self->{_nodes}->{"$kmer"}; + } + + return; +} + + +#### +sub purge_nodes_below_count { + my $self = shift; + + my ($min_count) = @_; + + my @nodes = $self->get_all_nodes(); + + my $num_deleted = 0; + + foreach my $node (@nodes) { + + if ($node->get_count() < $min_count) { + + my @colors = $node->get_colors(); + unless (scalar(@colors) == 1 && $colors[0] eq "black") { + next; + ## skipping base data (reads) below coverage limit + } + + $self->delete_node_from_graph($node); + + $num_deleted++; + } + } + + print STDERR " -purged nodes below count($min_count), deleted $num_deleted nodes.\n"; + + return; +} + + +#### +sub validate_graph { + my $self = shift; + + print STDERR "Validating graph.\n"; + + my @nodes = $self->get_all_nodes(); + + my %node_ids_in_graph; + + foreach my $node (@nodes) { + + my $node_id = $node->get_ID(); + $node_ids_in_graph{$node_id} = $node; + } + + foreach my $node (@nodes) { + + my $id = $node->get_ID(); + + my @prev_nodes = $node->get_all_prev_nodes(); + foreach my $prev_node (@prev_nodes) { + my $prev_id = $prev_node->get_ID(); + + unless (exists ($node_ids_in_graph{$prev_id})) { + print STDERR "Error, $prev_id exists as prev_node to $id, but not in graph\n"; + print STDERR $prev_node->get_sequence() . "\n" . $prev_node->toString(); + confess; + } + + unless ($prev_node->has_next_node($node)) { + confess "Error, prev_node: $prev_id lacks current next node $id as link.\n"; + } + + + } + + my @next_nodes = $node->get_all_next_nodes(); + foreach my $next_node (@next_nodes) { + my $next_id = $next_node->get_ID(); + + unless (exists ($node_ids_in_graph{$next_id})) { + print STDERR "Error, $next_id exists as next_node to $id, but not in graph\n"; + print STDERR $next_node->get_sequence() . "\n" . $next_node->toString(); + confess; + } + + unless ($next_node->has_prev_node($node)) { + confess "Error, next node: $next_id lacks current node $id as prev link.\n"; + } + + + } + + } + + return; +} + +#### +sub print_path { + my $self = shift; + my @nodes = @_; + + my $counter = 0; + my $prev_node; + + foreach my $node (@nodes) { + $counter++; + printf("%4s", $counter); + + my $edge_count = ""; + if ($prev_node) { + $edge_count = " edge_weight: " . $self->get_edge_count($prev_node, $node); + } + + print " " . $node->get_value() . " C:" . $node->get_count() . " " . join(",", $node->get_colors()) . "$edge_count\n"; + + $prev_node = $node; + + } + print "\n"; + + return; +} + + +sub _compatible_decorations { + my ($nodeA, $nodeB) = @_; + + + my @colorsA = $nodeA->get_colors(); + my @colorsB = $nodeB->get_colors(); + + unless (@colorsA || @colorsB) { + # no decorations. + return(1); + } + + my %counts; + foreach my $color (@colorsA, @colorsB) { + $counts{$color}++; + } + + my @colors_unique = grep { $counts{$_} == 1 } keys %counts; + + if (@colors_unique) { + #print STDERR "unique colors: @colors_unique\n"; + return(0); + } + + else { + return(1); + } +} + + + + +#### +sub score_nodes { + my $self = shift; + + $self->assign_depth_to_nodes(); + + ## score = sum prev counts for those reads that are consistent, excluding those that are from bifurcating reads. + + + my @nodes = sort {$a->{depth}<=>$b->{depth}} $self->get_all_nodes(); + + ## assign base scores: + foreach my $node (@nodes) { + + ## define base scores: contributions by reads that are not in prev or next nodes. + my %neighboring_reads = $self->gather_read_indices($node->get_all_prev_nodes(), $node->get_all_next_nodes()); + + my $base_score = 0; + + my $read_tracker = $node->get_ReadTracker(); + my @reads = $read_tracker->get_tracked_read_indices(); + foreach my $read (@reads) { + if ($node->{depth} == 0 || ! $neighboring_reads{$read}) { + my $count = $read_tracker->get_read_base_count($read); + + $base_score += $count; + #print "$node has read [$read] with count [$count]\n"; + + } + } + + $node->{_base_score} = $base_score; + $node->{_sum_score} = $base_score; # init + #print "$node : base score = $base_score\n"; + } + + + ## assign sum scores + foreach my $node (@nodes) { + my %indices_exclude; + + my $read_tracker = $node->get_ReadTracker(); + + my $highest_prev_score = $node->{_sum_score}; + my $highest_prev_node = undef; + + foreach my $prev_node ($node->get_all_prev_nodes()) { + ## want reads that are in current node and prev node, but not other next nodes of prev node (exclusive to current node). + my $sum_score = $prev_node->{_sum_score}; + + my @other_prev_next_nodes = grep { $_ ne $node } $prev_node->get_all_next_nodes(); + + my %reads_ignore = $self->gather_read_indices(@other_prev_next_nodes); + + foreach my $read ($read_tracker->get_tracked_read_indices()) { + + if (! $reads_ignore{$read}) { + + $sum_score += $read_tracker->get_read_base_count($read); + } + } + + push (@{$prev_node->{_forward_scores}}, { next_node => $node, + score => $sum_score, + } ); + + + if ($sum_score > $highest_prev_score) { + $highest_prev_score = $sum_score; + $highest_prev_node = $prev_node; + } + } + + $node->{_sum_score} = $highest_prev_score; + $node->{_best_prev} = $highest_prev_node; + } + + + ## find highest sum score + my $highest_sum_score = 0; + my $best_node = undef; + + foreach my $node (@nodes) { + #print "Node: $node, has sum score: " . $node->{_sum_score} . "\n"; + if ($node->{_sum_score} > $highest_sum_score) { + $highest_sum_score = $node->{_sum_score}; + $best_node = $node; + } + } + + + my @path_nodes; + my $prev_node = $best_node; + while ($prev_node) { + #print "Backtracked: $prev_node " . $prev_node->{_sum_score} . "\n"; + #print $prev_node->toString(); + push (@path_nodes, $prev_node); + $prev_node = $prev_node->{_best_prev}; + + } + + @path_nodes = reverse @path_nodes; + + return (@path_nodes); + + +} + + +#### +sub extract_sequence_from_path { + my $self = shift; + my (@ordered_nodes) = @_; # ordered left to right + + my @seqs; + + foreach my $node (@ordered_nodes) { + + $node->{_visited} = 1; + + push (@seqs, $node->get_sequence()); + + } + + + my $kmer_length = $self->{KmerLength}; + + my $final_seq = shift @seqs; ## first one gets full kmer treatment. + + foreach my $seq (@seqs) { + $seq = substr($seq, $kmer_length-1); + + $final_seq .= $seq; + } + + + return($final_seq); +} + + + +#### +sub gather_read_indices { + my $self = shift; + my (@nodes) = @_; + + my %indices; + + foreach my $node (@nodes) { + + my $read_tracker = $node->get_ReadTracker(); + + my @read_indices = $read_tracker->get_tracked_read_indices(); + + foreach my $read (@read_indices) { + $indices{$read}=1; + } + } + + return(%indices); +} + + +#### +sub assign_depth_to_nodes { + my $self = shift; + print STDERR "Assigning depth to nodes\n"; + + my @root_nodes = $self->get_root_nodes(); + + ## base cases. + foreach my $root (@root_nodes) { + $root->{depth} = 0; + } + + foreach my $node ($self->get_all_nodes()) { + if (! defined ($node->{depth})) { + $self->assign_depth($node, {}); + } + } + + + return; +} + + +#### +sub assign_depth { + my $self = shift; + my $node = shift; + my $seen_href = shift; + + my @ancestors = $node->get_all_prev_nodes(); + + my $max_depth = 0; + + foreach my $ancestor (@ancestors) { + + my $depth; + if (defined($ancestor->{depth})) { + $depth = 1 + $ancestor->{depth}; + } + else { + + if ($seen_href->{$ancestor}) { + # print "Breaking cycle: $ancestor\n" . join("\n", keys %$seen_href) . "\n"; + next; + } + $seen_href->{$ancestor} = 1; + + $depth = 1 + $self->assign_depth($ancestor, $seen_href); + } + if ($depth > $max_depth) { + $max_depth = $depth; + } + } + + # print "$node\tdepth: $max_depth\n"; + $node->{depth} = $max_depth; + return($max_depth); +} + + +#### +sub all_nodes_visited { + my $self = shift; + + my @nodes = $self->get_all_nodes(); + + foreach my $node (@nodes) { + + if (! $node->{_visited}) { + return(0); + } + } + + return(1); # all visited. +} + +#### +sub extract_path_from_unvisited_node { + my $self = shift; + + my @nodes = $self->get_all_nodes(); + + my $highest_scoring_unvisited_node = undef; + my $highest_score = -1; + + foreach my $node (@nodes) { + if (! $node->{_visited}) { + if ($node->{_base_score} > $highest_score) { + $highest_score = $node->{_base_score}; + $highest_scoring_unvisited_node = $node; + } + } + } + + unless ($highest_scoring_unvisited_node) { + confess "Error, no unvisited node found"; + } + + my @best_path; + + my %seen; # don't follow cycles. FIXME: cycles shouldn't be part of the scoring, though... bug somewhere upstream? + + my $prev_node = $highest_scoring_unvisited_node; + while ($prev_node && ! $prev_node->{_visited} && ! $seen{$prev_node}) { + print "walking prev_node: $prev_node\n" if $main::SEE; + $seen{$prev_node} = 1; + unshift(@best_path, $prev_node); + $prev_node = $prev_node->{_best_prev}; + } + + ## now track forward. + my $next_node = $highest_scoring_unvisited_node; + delete $seen{$next_node}; # already used above. + while ($next_node && ! $next_node->{_visited} && ! $seen{$next_node}) { + print "walking next_node: $next_node\n" if $main::SEE; + $seen{$next_node} = 1; + my @structs = @{$next_node->{_forward_scores}}; + if (@structs) { + @structs = sort {$a->{score}<=>$b->{score}} @structs; + my $best_struct = pop @structs; + push (@best_path, $next_node); + $next_node = $best_struct->{next_node}; + } + else { + $next_node = undef; + } + } + + return(@best_path); +} + + + +#### +sub prune_low_weight_edges { + my $self = shift; + my %params = @_; + + my $edge_weight_threshold = $params{edge_weight_threshold}; + + + my @nodes = $self->get_all_nodes(); + + my @edges_to_prune; + + foreach my $node (@nodes) { + my @prev_nodes = $node->get_all_prev_nodes(); + + foreach my $prev_node (@prev_nodes) { + + my $edge_ratio = $self->compute_edge_support_ratio($prev_node, $node); + + #print "EdgeRatio: $edge_ratio\n"; + + if ($edge_ratio < $edge_weight_threshold) { + push (@edges_to_prune, [$prev_node, $node]); + + } + } + } + + + if (@edges_to_prune) { + + foreach my $edges_to_prune (@edges_to_prune) { + + my ($prev_node, $node) = @$edges_to_prune; + $self->prune_edge($prev_node, $node); + } + + return(scalar(@edges_to_prune)); + } + else { + return(0); + } + +} + +#### +sub prune_singletons { + my $self = shift; + + my ($min_seq_length) = @_; + + my @nodes = $self->get_all_nodes(); + + foreach my $node (@nodes) { + + if ( (! $node->get_all_prev_nodes()) + && + (! $node->get_all_next_nodes()) ) { + + if (length ($node->get_sequence()) < $min_seq_length) { + + $self->delete_node_from_graph($node); + } + } + } + + return; +} + + +#### +sub compute_edge_support_ratio { + my $self = shift; + my ($prev_node, $node) = @_; + + ## ratio = reads in common / (all reads out of prev UNION all reads into node) + + my @reads_exiting_prev_node = $prev_node->get_reads_exiting_node(); + my %exiting_reads = map { + $_ => 1 } @reads_exiting_prev_node; + + my @reads_entering_node = $node->get_reads_entering_node(); + my %entering_reads = map { + $_ => 1 } @reads_entering_node; + + my %all_reads = map { + $_ => 1 } (@reads_exiting_prev_node, @reads_entering_node); + + my @all = keys %all_reads; + my @shared; + + foreach my $read (@all) { + if ($entering_reads{$read} && $exiting_reads{$read}) { + push (@shared, $read); + } + } + + my $ratio = scalar(@shared) / scalar(@all); + + return($ratio); +} + + + +#### +sub prune_dangling_nodes { + my $self = shift; + my %params = @_; + + my $min_leaf_node_length = $params{min_leaf_node_length}; + my $min_leaf_node_avg_cov = $params{min_leaf_node_avg_cov}; + + unless (defined ($min_leaf_node_length) + && + defined ($min_leaf_node_avg_cov) ) { + + confess "Error, must define values for both: min_leaf_node_length && min_leaf_node_avg_cov"; + } + + my @nodes = $self->get_all_nodes(); + + my @nodes_to_prune; + + my $kmer_length = $self->{KmerLength}; + + foreach my $node (@nodes) { + + ## check to see if it's a dangling node. + if ( (! $node->get_all_prev_nodes()) + || + (! $node->get_all_next_nodes()) + ) { + + my $length = length($node->get_sequence()); + + my $cov = $node->get_count(); + + my $avg_cov = $cov / ($length - ($kmer_length - 1)); + + if ($avg_cov < $min_leaf_node_avg_cov && $length < $min_leaf_node_length) { + push (@nodes_to_prune, $node); + } + + } + } + + if (@nodes_to_prune) { + + foreach my $node (@nodes_to_prune) { + + $self->prune_nodes_from_graph($node); + } + + return(scalar(@nodes_to_prune)); # nodes pruned. + } + + else { + return(0); # no nodes pruned. + } + +} + + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/KmerGraphLib/StringNode.pm b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/StringNode.pm new file mode 100644 index 0000000..1154d6f --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/KmerGraphLib/StringNode.pm @@ -0,0 +1,173 @@ +package StringNode; + +use strict; +use warnings; +use Carp; +use ReadTracker; + +use base qw (GenericNode); + +sub new { + my $packagename = shift; + my ($kmer_seq, $accession) = @_; + + my $self = $packagename->SUPER::new($kmer_seq); + + $self->{_count} = 0; + $self->{_ReadTracker} = new ReadTracker(); + + $self->{depth} = undef; + + ## for DP scans: + $self->{_visited} = 0; + $self->{_base_score} = 0; + $self->{_sum_score} = 0; + $self->{_best_prev} = undef; # prev_node in highest scoring path. + $self->{_forward_scores} = []; # stores structs of { node => ref, score => $score} + + + bless ($self, $packagename); + + return($self); +} + + +#### +sub get_ReadTracker { + my $self = shift; + return($self->{_ReadTracker}); +} + + +sub get_sequence { + my $self = shift; + + return($self->get_value()); +} + +sub set_sequence { + my $self = shift; + my $sequence = shift; + + unless ($sequence =~ /\w/) { + confess "Error, need sequence"; + } + + $self->set_value($sequence); + + return; +} + +sub track_reads { + my $self = shift; + my @accs = @_; + + $self->{_ReadTracker}->track_reads(@accs); + + return; +} + +sub get_count { + my $self = shift; + + return($self->{_count}); +} + +sub set_count { + my $self = shift; + + my ($count) = @_; + + $self->{_count} = $count; + return; +} + + + +#### +sub toString { + my $self = shift; + + my @prev_nodes = $self->get_all_prev_nodes(); + + my @next_nodes = $self->get_all_next_nodes(); + + my $text = ""; + + foreach my $prev_node (@prev_nodes) { + $text .= "P " . $prev_node->get_value() . "(" . $prev_node->get_count() . ") $prev_node " . join(",", $prev_node->get_colors()) . "\n"; + } + $text .= "X " . $self->get_value() . "(" . $self->get_count() . ") $self " . join(",", $self->get_colors()) . "\n"; + + foreach my $next_node (@next_nodes) { + $text .= "N " . $next_node->get_value() . "(" . $self->get_count() . ") $next_node " . join(",", $next_node->get_colors()) . "\n"; + } + + return($text); +} + + +#### +sub get_reads_exiting_node { + my $self = shift; + + my @next_nodes = $self->get_all_next_nodes(); + unless (@next_nodes) { + return(); + } + + my $read_tracker = $self->get_ReadTracker(); + + my @reads_in_node = $read_tracker->get_tracked_read_indices(); + + my @reads_in_next_nodes; + foreach my $next_node (@next_nodes) { + push (@reads_in_next_nodes, $next_node->get_ReadTracker()->get_tracked_read_indices()); + } + + my %next_node_reads = map { + $_ => 1 } @reads_in_next_nodes; + + my @exiting_reads; + foreach my $read (@reads_in_node) { + if ($next_node_reads{$read}) { + push (@exiting_reads, $read); + } + } + + return(@exiting_reads); +} + +#### +sub get_reads_entering_node { + my $self = shift; + + my @prev_nodes = $self->get_all_prev_nodes(); + unless (@prev_nodes) { + return(); + } + + my $read_tracker = $self->get_ReadTracker(); + + my @reads_in_node = $read_tracker->get_tracked_read_indices(); + + my @reads_in_prev_nodes; + foreach my $prev_node (@prev_nodes) { + push (@reads_in_prev_nodes, $prev_node->get_ReadTracker()->get_tracked_read_indices()); + } + + my %prev_node_reads = map { + $_ => 1 } @reads_in_prev_nodes; + + my @entering_reads; + foreach my $read (@reads_in_node) { + if ($prev_node_reads{$read}) { + push (@entering_reads, $read); + } + } + + return(@entering_reads); +} + + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/Ktree.pm b/99.scripts/trinity_utils/PerlLib/Ktree.pm new file mode 100644 index 0000000..cca72fa --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Ktree.pm @@ -0,0 +1,156 @@ +package Ktree; + +use strict; +use warnings; +use Carp; + +sub new { + my $packagename = shift; + + my $self = { _root => KtreeNode->new("", 0) }; + + bless ($self, $packagename); + + return($self); +} + +sub add_kmer { + my $self = shift; + my ($kmer) = @_; + + + unless (defined $kmer) { + confess "error, require param kmer"; + } + + my $root_node = $self->{_root}; + + my @seq = split(//, $kmer); + + my $node = $root_node; + do { + my $char = shift @seq; + $node = $node->get_child($char); + } while (@seq); + + $node->set_val( $node->get_val() + 1 ); + + return; +} + +sub report_kmer_counts { + my $self = shift; + + my $root_node = $self->{_root}; + + &_recurse_through_kmer_counts("", $root_node); + + return; +} + +sub _recurse_through_kmer_counts { + my ($prefix, $node) = @_; + + my $char = $node->get_char(); + + my @children_chars = $node->get_children_chars(); + + if (@children_chars) { + foreach my $child_char (@children_chars) { + my $child_node = $node->get_child($child_char); + &_recurse_through_kmer_counts($prefix . $char, $child_node); + } + } + else { + # base case + my $val = $node->get_val(); + print join("\t", $prefix . $char, $val) . "\n"; + } + + return; +} + + + +package KtreeNode; + +use strict; +use warnings; +use Carp; + + +sub new { + my $packagename = shift; + my ($char, $val) = @_; + + unless (defined $char && defined $val) { + confess "Error, require (character, val) as parameter"; + } + + my $self = { char => $char, + val => $val, + children => {}, + }; + + bless ($self, $packagename); + + return($self); +} + + + + +#### +sub get_child { + my $self = shift; + my ($char) = @_; + + unless (defined $char) { + confess "error, parameter 'char' required"; + } + + my $child = $self->{children}->{$char}; + unless (ref $child) { + + $child = $self->{children}->{$char} = new KtreeNode($char, 0); + } + + return($child); +} + + +sub get_children_chars { + my $self = shift; + + my @chars = keys %{$self->{children}}; + return(@chars); +} + + +sub get_char { + my $self = shift; + return($self->{char}); +} + + +#### +sub get_val { + my $self = shift; + return($self->{val}); +} + +#### +sub set_val { + my $self = shift; + my $val = shift; + unless (defined $val) { + confess "error, require val as param"; + } + + $self->{val} = $val; + + return; +} + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/Longest_orf.pm b/99.scripts/trinity_utils/PerlLib/Longest_orf.pm new file mode 100644 index 0000000..70177aa --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Longest_orf.pm @@ -0,0 +1,371 @@ +#!/usr/local/bin/perl + +package main; +our $SEE; + +package Longest_orf; + +use strict; +use warnings; +use Nuc_translator; +use Carp; + +## if allow_partials is set, partial orfs are included in the analysis. + +# below used to be static, now instance vars. +#my $ALLOW_5PRIME_PARTIALS = 0; #allow for lacking start codon in logest orf. +#my $ALLOW_3PRIME_PARTIALS = 0; #allow for lacking stop codon in longest orf. +#my $FORWARD_STRAND = 1; #default set to true (analyze forward strand) +#my $REVERSE_STRAND = 1; #default set to true. +#my $ALLOW_NON_MET_STARTS = 0; #allow for non-methionine start codons. + + +sub new { + shift; + + ## This object stores the longest ORF identified. + my @stop_codons = &Nuc_translator::get_stop_codons(); # live call, depends on current genetic code. + print "Stop codons in use: @stop_codons, set dynamically via current Nuc_translator settings.\n" if $SEE; + unless (@stop_codons) { + confess "Fatal, no stop codons set"; + } + + my $obj = { pep_seq => undef, + nt_seq => undef, + length => undef, #length of nt_seq + end5 => undef, + end3 => undef, + all_ORFS=>[], #container holds all ORFs found in order of decreasing length. Use orfs() method to retrieve them. + stop_codons => [@stop_codons], + + ## ORF settings + ALLOW_5PRIME_PARTIALS => 0, + ALLOW_3PRIME_PARTIALS => 0, + FORWARD_STRAND => 1, + REVERSE_STRAND => 1, + ALLOW_NON_MET_STARTS => 0 + + }; + bless ($obj); + return ($obj); +} + +## can include partial orfs at end of sequence. +sub allow_partials { + my $self = shift; + die unless (ref $self); + $self->{ALLOW_5PRIME_PARTIALS} = 1; + $self->{ALLOW_3PRIME_PARTIALS} = 1; + + if ($SEE) { + print "Longest_orf: allowing both 5' and 3' partials.\n"; + } +} + +sub allow_5prime_partials { + my $self = shift; + die unless (ref $self); + $self->{ALLOW_5PRIME_PARTIALS} = 1; + if ($SEE) { + print "Longest_orf: allowing 5prime partials.\n"; + } +} + +sub allow_3prime_partials { + my $self = shift; + die unless (ref $self); + $self->{ALLOW_3PRIME_PARTIALS} = 1; + if ($SEE) { + print "Longest_orf: allowing 3prime partials\n"; + } +} + +sub forward_strand_only { + my $self = shift; + die unless (ref $self); + $self->{REVERSE_STRAND} = 0; + if ($SEE) { + print "Longest_orf: forward strand only.\n"; + } + +} + +sub reverse_strand_only { + my $self = shift; + die unless (ref $self); + $self->{FORWARD_STRAND} = 0; + if ($SEE) { + print "Longest_orf: reverse strand only.\n"; + } + +} + +sub allow_non_met_starts { + my $self = shift; + $self->{ALLOW_NON_MET_STARTS} = 1; + if ($SEE) { + print "Longest_orf: allowing non Met start codons.\n"; + } +} + + +sub get_longest_orf { + my $self = shift; + my $input_sequence = shift; + + unless ($input_sequence) { + confess "I require a cDNA nucleotide sequence as my only parameter "; + return; + } + unless (length ($input_sequence) >= 3) { + print STDERR "Sequence must code for at least a codon. Your seq_length is too short\n"; + return; + } + my @orfList = $self->capture_all_ORFs($input_sequence); + # print "Found " . scalar @orfList . " orfs.\n"; + if (@orfList) { + return ($orfList[0]); # longest ORF found is first in the sorted list. + } + else { + ## no ORFs found + return (undef); + } +} + + +sub capture_all_ORFs { + + my $self = shift; + my $input_sequence = shift; + + unless ($input_sequence) { + confess "I require a cDNA nucleotide sequence as my only parameter\n"; + return; + } + unless (length ($input_sequence) >= 3) { + print STDERR "Sequence must code for at least a codon. Your seq_length is too short\n"; + return; + } + + $input_sequence = lc ($input_sequence); + + my (@starts, @stops, @orfs); + + if ($self->{FORWARD_STRAND}) { + ## analyse forward position + @stops = $self->identify_putative_stops($input_sequence); + @starts = $self->identify_putative_starts($input_sequence,\@stops); + @orfs = $self->get_orfs (\@starts, \@stops, $input_sequence, '+'); + } + + if ($self->{REVERSE_STRAND}) { + ## reverse complement sequence and do again + $input_sequence = &revcomp ($input_sequence); + @stops = $self->identify_putative_stops($input_sequence); + @starts = $self->identify_putative_starts($input_sequence, \@stops); + push (@orfs, $self->get_orfs (\@starts, \@stops, $input_sequence, '-')); + } + + if (@orfs) { + ## set in order of decreasing length + @orfs = reverse sort {$a->{length} <=> $b->{length}} @orfs; + + my $longest_orf = $orfs[0]; + my $start = $longest_orf->{start}; + my $stop = $longest_orf->{stop}; + my $seq = $longest_orf->{sequence}; + my $length = length($seq); + my $protein = &translate_sequence($seq, 1); + $self->{end5} = $start; ## now coord is seq_based instead of array based. + $self->{end3} = $stop; + $self->{length} = $length; + $self->{nt_seq} = $seq; + $self->{pep_seq} = $protein; + $self->{all_ORFS} = \@orfs; + } + + return (@orfs); +} + +sub orfs { + my $self = shift; + return (@{$self->{all_ORFS}}); +} + +##################### +# supporting methods +##################### + +sub get_end5_end3 { + my $self = shift; + return ($self->{end5}, $self->{end3}); +} + +sub get_peptide_sequence { + my $self = shift; + return ($self->{pep_seq}); +} + +sub get_nucleotide_sequence { + my $self = shift; + return ($self->{nt_seq}); +} + + +sub toString { + my $self = shift; + my ($end5, $end3) = $self->get_end5_end3(); + my $protein = $self->get_peptide_sequence(); + my $nt_seq = $self->get_nucleotide_sequence(); + my $ret_string = "Coords: $end5, $end3\n" + . "Protein: $protein\n" + . "Nucleotides: $nt_seq\n"; + return ($ret_string); +} + + +################################# + +#Private methods: + + +sub get_orfs { + my ($self, $starts_ref, $stops_ref, $seq, $direction) = @_; + + unless ($starts_ref && $stops_ref && $seq && $direction) { + confess "Error, params not appropriate"; + } + + my %last_delete_pos = ( 0=>-1, + 1=>-1, + 2=>-1); #store position of last chosen stop codon in spec reading frame. + my @orfs; + my $seq_length = length ($seq); + + if ($SEE) { + print "Potential Start codons: " . join (", ", @$starts_ref) . "\n"; + print "Potential Stop codons: " . join (", ", @$stops_ref) . "\n"; + } + + + foreach my $start_pos (@{$starts_ref}) { + my $start_pos_frame = $start_pos % 3; + foreach my $stop_pos (@{$stops_ref}) { + # print "Comparing start: $start_pos to stop: $stop_pos, $direction\n"; + if ( ($stop_pos > $start_pos) && #end3 > end5 + ( ($stop_pos - $start_pos) % 3 == 0) #must be in-frame + && ($start_pos > $last_delete_pos{$start_pos_frame})) #only count each stop once. + { + + $last_delete_pos{$start_pos_frame} = $stop_pos; + my ($start_pos_adj, $stop_pos_adj) = ( ($start_pos+1), ($stop_pos+1+2)); + #print "Startposadj: $start_pos_adj\tStopPosadj: $stop_pos_adj\n"; + # sequence based position rather than array-based + + my ($start, $stop) = ($direction eq '+') ? ($start_pos_adj, $stop_pos_adj) + : (&revcomp_coord($start_pos_adj, $seq_length), &revcomp_coord($stop_pos_adj, $seq_length)); + + print "Retrieving ORF, Start: $start\tStop: $stop\n" if $SEE; + my $orfSeq = substr ($seq, $start_pos, ($stop_pos - $start_pos + 3)); #include the stop codon too. + my $protein = &translate_sequence($orfSeq, 1); + if ($protein =~ /\*.*\*/) { + confess "Fatal Error: Longest_orf: ORF returned which contains intervening stop(s): ($start-$stop, $direction\nProtein:\n$protein\nOf Nucleotide Seq:\n$seq\n"; + } + my $orf = { sequence => $orfSeq, + protein => $protein, + start=>$start, + stop=>$stop, + length=>length($orfSeq), + orient=>$direction + }; + push (@orfs, $orf); + last; + } + } + } + return (@orfs); +} + + +sub identify_putative_starts { + my ($self, $seq, $stops_aref) = @_; + my %starts; + my %stops; + foreach my $stop (@$stops_aref) { + $stops{$stop} = 1; + } + + if ($self->{ALLOW_5PRIME_PARTIALS} || $self->{ALLOW_NON_MET_STARTS}) { + $starts{0} = 1 unless $stops{0}; + $starts{1} = 1 unless $stops{1}; + $starts{2} = 1 unless $stops{2}; + } + + if (! $self->{ALLOW_NON_MET_STARTS}) { #Look for ATG start codons. + my $start_pos = index ($seq, "atg"); + while ($start_pos != -1) { + $starts{$start_pos} = 1; + #print "Start: $start_pos\n"; + $start_pos = index ($seq, "atg", ($start_pos + 1)); + } + } else { + # find all residues just subsequent to a stop codon, in-frame: + foreach my $stop (@$stops_aref) { + my $candidate_non_met_start = $stop +3; + unless ($stops{$candidate_non_met_start}) { + $starts{$candidate_non_met_start} = 1; + } + } + } + my @starts = sort {$a<=>$b} keys %starts; + return (@starts); +} + + +sub identify_putative_stops { + my ($self, $seq) = @_; + my %stops; + if ($self->{ALLOW_3PRIME_PARTIALS}) { + ## count terminal 3 nts as possible ORF terminators. + my $seq_length = length ($seq); + $stops{$seq_length} = 1; + $seq_length--; + $stops{$seq_length} = 1; + $seq_length--; + $stops{$seq_length} = 1; + } + my @stop_codons = @{$self->{stop_codons}}; + foreach my $stop_codon (@stop_codons) { + $stop_codon = lc $stop_codon; + print "Searching for stop codon: ($stop_codon).\n" if $SEE; + my $stop_pos = index ($seq, $stop_codon); + while ($stop_pos != -1) { + $stops{$stop_pos} = 1; + $stop_pos = index ($seq, $stop_codon, ($stop_pos + 1)); #include the stop codon too. + } + } + my @stops = sort {$a<=>$b} keys %stops; + return (@stops); +} + + +sub revcomp { + my ($seq) = @_; + my $reversed_seq = reverse ($seq); + $reversed_seq =~ tr/ACGTacgtyrkm/TGCAtgcarymk/; + return ($reversed_seq); +} + + +sub revcomp_coord { + my ($coord, $seq_length) = @_; + return ($seq_length - $coord + 1); +} + + + + + +1; + + diff --git a/99.scripts/trinity_utils/PerlLib/Nuc_translator.pm b/99.scripts/trinity_utils/PerlLib/Nuc_translator.pm new file mode 100644 index 0000000..a45c063 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Nuc_translator.pm @@ -0,0 +1,647 @@ +#!/usr/bin/env perl + +package main; +our $SEE; + +package Nuc_translator; + +use strict; +require Exporter; +use Carp; + + +our @ISA = qw (Exporter); +our @EXPORT = qw (get_genetic_codes translate_sequence get_protein reverse_complement); + +use vars qw ($currentCode %codon_table $init_codon_table_subref $support_Thymine_and_Uracil_subref); + + +=head1 NAME + +package Nuc_translator.pm + + +=head1 SYNOPSIS + +Nuc_translator::use_specified_genetic_code ("universal"); + +my $nuc_sequence = "atgaaagggccctga"; + +my $translation_frame = 1; + +my $protein = &translate_sequence($nuc_sequence, $translation_frame); + + +=head1 DESCRIPTION + +Methods are provided to translate nucleotide sequences into protein sequences using a specified genetic code. + +Available genetic codes include universal, Euplotes, Tetrahymena, Candida, Acetabularia + +For info on these codes, visit: + +http://golgi.harvard.edu/biolinks/gencode.html +(https://web.archive.org/web/20040216020103/http://golgi.harvard.edu/biolinks/gencode.html) + +Methods exported by this package include: + +translate_sequence() + +get_protein() + +reverse_complement() + +To change the translation code, the following fully qualified method must be used: + +Nuc_translator::use_specified_genetic_code() + + +=head1 Methods + + +=cut + + + +## See http://golgi.harvard.edu/biolinks/gencode.html +my %SUPPORTED_GENETIC_CODES = ( Universal => 1, + + Tetrahymena => 1, + Acetabularia => 1, + Ciliate => 1, + Dasycladacean => 1, + Hexamita => 1, + + Candida => 1, + + Euplotid => 1, + + SR1_Gracilibacteria => 1, + + Pachysolen_tannophilus => 1, + + Mesodinium => 1, + + Peritrich => 1, + + + + 'Mitochondrial-Vertebrates' => 1, + 'Mitochondrial-Yeast' => 1, + 'Mitochondrial-Invertebrates' => 1, + "Mitochondrial-Protozoan" => 1, + "Mitochondrial-Echinoderm" => 1, + "Mitochondrial-Ascidian" => 1, + "Mitochondrial-Flatworm" => 1, + "Mitochondrial-Chlorophycean" => 1, + "Mitochondrial-Trematode" => 1, + "Mitochondrial-Scenedesmus_obliquus" => 1, + "Mitochondrial-Thraustochytrium" => 1, + "Mitochondrial-Pterobranchia" => 1, + + ); + + + + + +=over 4 + +=item get_genetic_codes() + +B provides the list of supported genetic codes + +B + +B list of genetic codes + +=back + +=cut + +#### +sub get_genetic_codes { + return (sort keys %SUPPORTED_GENETIC_CODES); +} + + + +#### +sub show_translation_table { + my ($genetic_code) = @_; + + &use_specified_genetic_code($genetic_code); + + my @nucs = qw(T C A G); + + + my $translation_line = ""; + my $c1_line = ""; + my $c2_line = ""; + my $c3_line = ""; + + for my $c1 (@nucs) { + for my $c2 (@nucs) { + for my $c3 (@nucs) { + $c1_line .= $c1; + $c2_line .= $c2; + $c3_line .= $c3; + + my $codon = $c1 . $c2 . $c3; + my $translation = $codon_table{$codon}; + $translation_line .= "$translation"; + } + } + } + + my $translation_table = join("\n", $translation_line, $c1_line, $c2_line, $c3_line) . "\n"; + + return($translation_table); +} + + +=over 4 + +=item translate_sequence() + +B translates a nucleotide sequence given a specific frame 1-6. + +B $nuc_sequence, $frame + +B $protein_sequence + +=back + +=cut + + + +sub translate_sequence { + my ($sequence, $frame) = @_; + + $sequence = uc ($sequence); + $sequence =~ tr/U/T/; + my $seq_length = length ($sequence); + unless ($frame >= 1 and $frame <= 6) { + confess "Frame $frame is not allowed. Only between 1 and 6"; + } + + if ($frame > 3) { + # on reverse strand. Revcomp the sequence and reset the frame + $sequence = &reverse_complement($sequence); + if ($frame == 4) { + $frame = 1; + } + elsif ($frame == 5) { + $frame = 2; + } + elsif ($frame == 6) { + $frame = 3; + } + } + + $sequence =~ tr/T/U/; + my $start_point = $frame - 1; + my $protein_sequence; + for (my $i = $start_point; $i < $seq_length; $i+=3) { + my $codon = substr($sequence, $i, 3); + my $amino_acid; + if (exists($codon_table{$codon})) { + $amino_acid = $codon_table{$codon}; + } else { + if (length($codon) == 3) { + $amino_acid = 'X'; + } else { + $amino_acid = ""; + } + } + $protein_sequence .= $amino_acid; + } + return($protein_sequence); +} + + + +=over 4 + +=item get_protein() + +B translates nucleotide sequence into a protein sequence. All 3 forward translation frames are tried +and the first reading frame found to translate without stop codons is returned. If all 3 frames provide stop codons, the protein with the least number of stops is returned. + +B $nucleotide_sequence + +B $protein_sequence + +=back + +=cut + + + +sub get_protein { + my ($sequence) = @_; + + ## Assume frame 1 unless multiple stops appear. + my $least_stops = undef(); + my $least_stop_prot_seq = ""; + foreach my $forward_frame (1, 2, 3) { + my $protein = &translate_sequence($sequence, $forward_frame); + my $num_stops = &count_stops_in_prot_seq($protein); + if ($num_stops == 0) { + return ($protein); + } else { + if (!defined($least_stops)) { + #initialize data + $least_stops = $num_stops; + $least_stop_prot_seq = $protein; + } elsif ($num_stops < $least_stops) { + $least_stops = $num_stops; + $least_stop_prot_seq = $protein; + } else { + #keeping original $num_stops and $least_stop_prot_seq + } + } + } + return ($least_stop_prot_seq); +} + + +=over 4 + +=item reverse_complement() + +B reverse complements a nucleotide sequence + +B $nucleotide_sequence + +B $nucleotide_sequence_rev_comped + +=back + +=cut + + + +sub reverse_complement { + my($s) = @_; + my ($rc); + $rc = reverse ($s); + $rc =~tr/ACGTacgtyrkmYRKMUu/TGCAtgcarymkRYMKAa/; + return($rc); +} + + +#### +sub count_stops_in_prot_seq { + my ($prot_seq) = @_; + chop $prot_seq; #remove trailing stop. + my $stop_num = 0; + while ($prot_seq =~ /\*/g) { + $stop_num++; + } + return ($stop_num); +} + +#### +sub use_specified_genetic_code { + my ($special_code) = @_; + print STDERR "using special genetic code $special_code\n" if $SEE; + unless ($SUPPORTED_GENETIC_CODES{$special_code}) { + die "Sorry, $special_code is not currently supported or recognized.\n"; + } + &$init_codon_table_subref(); ## Restore default universal code. Others are variations on this. + $currentCode = $special_code; + + if ($special_code eq "Universal") { + # already set. + } + + elsif ($special_code =~ /^(Tetrahymena|Acetabularia|Ciliate|Dasycladacean|Hexamita)$/) { + $codon_table{UAA} = "Q"; + $codon_table{UAG} = "Q"; + } + + elsif ($special_code eq "Candida") { + $codon_table{CUG} = "S"; + } + + elsif ($special_code eq "Euplotid") { + $codon_table{UGA} = "C"; # not * + } + + elsif ($special_code eq "SR1_Gracilibacteria") { + $codon_table{UGA} = "G"; # not * + } + + elsif ($special_code eq "Pachysolen_tannophilus") { + $codon_table{CUG} = "A"; # not L + } + elsif ($special_code eq "Mesodinium") { + $codon_table{UAA} = "Y"; # not * + $codon_table{UAG} = "Y"; # not * + } + + elsif ($special_code eq "Peritrich") { + $codon_table{UAA} = "E"; # not * + $codon_table{UAG} = "E"; # not * + } + + elsif ($special_code =~ /Mitochondrial/) { + &_set_mitochondrial_code($special_code); + } + + else { + ## shouldn't ever get here anyway. + confess "Error, code $special_code is not recognized.\n"; + } + + + &$support_Thymine_and_Uracil_subref(); + +} + + +#### +sub _set_mitochondrial_code { + my $code = shift; + + # see: https://www.ncbi.nlm.nih.gov/Taxonomy/Utils/wprintgc.cgi#SG2 + + if ($code eq "Mitochondrial-Vertebrates") { + $codon_table{UGA} = "W"; + $codon_table{AUA} = "M"; + $codon_table{AGA} = "*"; + $codon_table{AGG} = "*"; + } + elsif ($code eq "Mitochondrial-Yeast") { + $codon_table{AUA} = "M"; # instead of I + $codon_table{CUU} = $codon_table{CUC} = $codon_table{CUA} = $codon_table{CUG} = "T"; # instead of L + $codon_table{UGA} = "W"; # instead of * + } + elsif ($code eq "Mitochondrial-Invertebrates") { + $codon_table{AGA} = "S"; + $codon_table{AGG} = "S"; + $codon_table{AUA} = "M"; + $codon_table{UGA} = "W"; + } + elsif ($code eq "Mitochondrial-Protozoan") { + $codon_table{UGA} = "W"; + } + elsif ($code eq "Mitochondrial-Echinoderm" || $code eq "Mitochondrial-Flatworm") { + $codon_table{AAA} = "N"; # not K + $codon_table{AGA} = "S"; # not R + $codon_table{AGG} = "S"; # not R + $codon_table{UGA} = "W"; # not * + } + elsif ($code eq "Mitochondrial-Ascidian") { + $codon_table{AGA} = "G"; # not R + $codon_table{AGG} = "G"; # not R + $codon_table{AUA} = "M"; # not I + $codon_table{UGA} = "W"; # not * + } + + elsif ($code eq "Mitochondrial-Chlorophycean") { + $codon_table{UAG} = "L"; # not * + } + elsif ($code eq "Mitochondrial-Trematode") { + $codon_table{UGA} = "W"; # not * + $codon_table{AUA} = "M"; # not I + $codon_table{AGA} = "S"; # not R + $codon_table{AGG} = "S"; # not R + $codon_table{AAA} = "N"; # not K + } + elsif ($code eq "Mitochondrial-Scenedesmus_obliquus") { + $codon_table{UCA} = "*"; # not S + $codon_table{UAG} = "L"; # not * + } + elsif ($code eq "Mitochondrial-Thraustochytrium") { + $codon_table{UUA} = "*"; + } + elsif ($code eq "Mitochondrial-Pterobranchia") { + $codon_table{AGA} = "S"; # not R + $codon_table{AGG} = "K"; # not R + $codon_table{UGA} = "W"; # not * + } + + + else { + confess "Sorry, $code hasn't been fully implemented yet."; + } + + + + + return; +} + + + +#### +sub get_stop_codons { + my @stop_codons; + foreach my $codon (keys %codon_table) { + if ($codon_table{$codon} eq '*') { + push (@stop_codons, $codon); + } + } + #foreach my $codon (@stop_codons) { + # $codon =~ tr/U/T/; + #} + return (@stop_codons); +} + + +BEGIN { + $init_codon_table_subref = sub { + print STDERR "initing codon table.\n" if $SEE; + ## Set to Universal Genetic Code + $currentCode = "universal"; + + %codon_table = ( UUU => 'F', + UUC => 'F', + UUA => 'L', + UUG => 'L', + + CUU => 'L', + CUC => 'L', + CUA => 'L', + CUG => 'L', + + AUU => 'I', + AUC => 'I', + AUA => 'I', + AUG => 'M', + + GUU => 'V', + GUC => 'V', + GUA => 'V', + GUG => 'V', + + UCU => 'S', + UCC => 'S', + UCA => 'S', + UCG => 'S', + + CCU => 'P', + CCC => 'P', + CCA => 'P', + CCG => 'P', + + ACU => 'T', + ACC => 'T', + ACA => 'T', + ACG => 'T', + + GCU => 'A', + GCC => 'A', + GCA => 'A', + GCG => 'A', + + UAU => 'Y', + UAC => 'Y', + UAA => '*', + UAG => '*', + + CAU => 'H', + CAC => 'H', + CAA => 'Q', + CAG => 'Q', + + AAU => 'N', + AAC => 'N', + AAA => 'K', + AAG => 'K', + + GAU => 'D', + GAC => 'D', + GAA => 'E', + GAG => 'E', + + UGU => 'C', + UGC => 'C', + UGA => '*', + UGG => 'W', + + CGU => 'R', + CGC => 'R', + CGA => 'R', + CGG => 'R', + + AGU => 'S', + AGC => 'S', + AGA => 'R', + AGG => 'R', + + GGU => 'G', + GGC => 'G', + GGA => 'G', + GGG => 'G' + + ); + }; + + + $support_Thymine_and_Uracil_subref = sub { + + my @codons = keys %codon_table; + foreach my $codon (@codons) { + if ($codon =~ /U/) { + my $aa = $codon_table{$codon}; + my $T_codon = $codon; + $T_codon =~ s/U/T/g; + $codon_table{$T_codon} = $aa; + } + } + }; + + + # init codon table, using uracil codons + &$init_codon_table_subref(); + + # update codon table to also support thymine + &$support_Thymine_and_Uracil_subref(); +} + + + + +#### +sub run_test() { + my %expected_translation_tables = ( + 'Universal' => "FFLLSSSSYY**CC*WLLLLPPPPHHQQRRRRIIIMTTTTNNKKSSRRVVVVAAAADDEEGGGG", + 'Mitochondrial-Vertebrates' => "FFLLSSSSYY**CCWWLLLLPPPPHHQQRRRRIIMMTTTTNNKKSS**VVVVAAAADDEEGGGG", + 'Mitochondrial-Yeast' => "FFLLSSSSYY**CCWWTTTTPPPPHHQQRRRRIIMMTTTTNNKKSSRRVVVVAAAADDEEGGGG", + 'Mitochondrial-Protozoan' => "FFLLSSSSYY**CCWWLLLLPPPPHHQQRRRRIIIMTTTTNNKKSSRRVVVVAAAADDEEGGGG", + 'Mitochondrial-Invertebrates' => "FFLLSSSSYY**CCWWLLLLPPPPHHQQRRRRIIMMTTTTNNKKSSSSVVVVAAAADDEEGGGG", + 'Ciliate' => 'FFLLSSSSYYQQCC*WLLLLPPPPHHQQRRRRIIIMTTTTNNKKSSRRVVVVAAAADDEEGGGG', + 'Mitochondrial-Echinoderm' => "FFLLSSSSYY**CCWWLLLLPPPPHHQQRRRRIIIMTTTTNNNKSSSSVVVVAAAADDEEGGGG", + 'Euplotid' => "FFLLSSSSYY**CCCWLLLLPPPPHHQQRRRRIIIMTTTTNNKKSSRRVVVVAAAADDEEGGGG", + 'Candida' => "FFLLSSSSYY**CC*WLLLSPPPPHHQQRRRRIIIMTTTTNNKKSSRRVVVVAAAADDEEGGGG", + 'Mitochondrial-Ascidian' => "FFLLSSSSYY**CCWWLLLLPPPPHHQQRRRRIIMMTTTTNNKKSSGGVVVVAAAADDEEGGGG", + 'Mitochondrial-Chlorophycean' => "FFLLSSSSYY*LCC*WLLLLPPPPHHQQRRRRIIIMTTTTNNKKSSRRVVVVAAAADDEEGGGG", + 'Mitochondrial-Trematode' => "FFLLSSSSYY**CCWWLLLLPPPPHHQQRRRRIIMMTTTTNNNKSSSSVVVVAAAADDEEGGGG", + 'Mitochondrial-Scenedesmus_obliquus' => "FFLLSS*SYY*LCC*WLLLLPPPPHHQQRRRRIIIMTTTTNNKKSSRRVVVVAAAADDEEGGGG", + 'Mitochondrial-Thraustochytrium' => "FF*LSSSSYY**CC*WLLLLPPPPHHQQRRRRIIIMTTTTNNKKSSRRVVVVAAAADDEEGGGG", + 'Mitochondrial-Pterobranchia' => "FFLLSSSSYY**CCWWLLLLPPPPHHQQRRRRIIIMTTTTNNKKSSSKVVVVAAAADDEEGGGG", + 'SR1_Gracilibacteria' => "FFLLSSSSYY**CCGWLLLLPPPPHHQQRRRRIIIMTTTTNNKKSSRRVVVVAAAADDEEGGGG", + 'Pachysolen_tannophilus' => "FFLLSSSSYY**CC*WLLLAPPPPHHQQRRRRIIIMTTTTNNKKSSRRVVVVAAAADDEEGGGG", + 'Mesodinium' => "FFLLSSSSYYYYCC*WLLLLPPPPHHQQRRRRIIIMTTTTNNKKSSRRVVVVAAAADDEEGGGG", + 'Peritrich' => "FFLLSSSSYYEECC*WLLLLPPPPHHQQRRRRIIIMTTTTNNKKSSRRVVVVAAAADDEEGGGG", + ); + + my $exit_code = 0; + + foreach my $genetic_code (sort keys %expected_translation_tables) { + my $expected_translation = $expected_translation_tables{$genetic_code}; + + my $reconstructed_translation_table = &show_translation_table($genetic_code); + my @pts = split(/\n/, $reconstructed_translation_table); + my $trans_table = shift @pts; + + if ($trans_table eq $expected_translation) { + print "$genetic_code\tOK\n"; + } + else { + print "\n$genetic_code\tERROR:\n" + . "$expected_translation (EXPECTED)\n" + . "$trans_table (Computed)\n\n"; + + $exit_code = 1; + + } + + } + + exit($exit_code); + +} + +unless (caller) { + + my $genetic_code = $ARGV[0]; + + if ($genetic_code eq "TEST") { + &run_test(); + } + + if ($genetic_code) { + print "\nTranslation table for $genetic_code:\n\n"; + print &show_translation_table($genetic_code) . "\n\n"; + } + else { + print "Supported genetic codes:\n\n" . join("\n", &get_genetic_codes()) . "\n\n" + . "Choose one to show translation table.\n\n"; + } + + exit(0); + +} + + + +1; #end of module + + + + diff --git a/99.scripts/trinity_utils/PerlLib/Overlap_info.pm b/99.scripts/trinity_utils/PerlLib/Overlap_info.pm new file mode 100644 index 0000000..cb9d24b --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Overlap_info.pm @@ -0,0 +1,320 @@ +#!/usr/bin/env perl + +package Overlap_info; + +use strict; +use warnings; +use List::Util qw (min max); +use Carp; +use Data::Dumper; + + +## works on coordinate pairs: [a1, a2], [b1, b2] +sub overlap { + my ($coordsA_aref, $coordsB_aref) = @_; + + my ($lendA, $rendA) = sort {$a<=>$b} @$coordsA_aref; + + my ($lendB, $rendB) = sort {$a<=>$b} @$coordsB_aref; + + if ($lendA <= $rendB && $rendA >= $lendB) { + return(1); + } + else { + return(0); + } +} + + +sub contains { + my ($larger, $smaller) = @_; + + my ($smaller_lend, $smaller_rend) = sort {$a<=>$b} @$smaller; + my ($larger_lend, $larger_rend) = sort {$a<=>$b} @$larger; + + if ($smaller_lend >= $larger_lend && $smaller_rend <= $larger_rend) { + return(1); + } + + else { + #print "no containment: [$larger_lend-$larger_rend] no containment of [$smaller_lend-$smaller_rend]\n"; + return(0); + } +} + + +## works on coordinate pairs: [a1, a2], [b1, b2] +sub overlap_length { + my ($coordsA_aref, $coordsB_aref) = @_; + + if (&overlap($coordsA_aref, $coordsB_aref)) { + + my ($lendA, $rendA) = sort {$a<=>$b} @$coordsA_aref; + + my ($lendB, $rendB) = sort {$a<=>$b} @$coordsB_aref; + + if ($lendA > $lendB) { + # swap em + ($lendA, $rendA, $lendB, $rendB) = ($lendB, $rendB, $lendA, $rendA); + } + + my $overlap_lend = max($lendA, $lendB); + my $overlap_rend = min($rendA, $rendB); + + my $overlap_len = $overlap_rend - $overlap_lend + 1; + return($overlap_len); + + } + else { + return(0); + } +} + + + +## works on sets of coordinates: ([ [a1,a2], [a3,a4], ...]) , ([ [b1,b2], [b3,b4], ... ]) +sub sum_overlaps { + my ($coordsets_A_aref, $coordsets_B_aref) = @_; + + + ## Be sure that no intra-set coordinates overlap each other. TODO: force this sanity check + + my $sum = 0; + + foreach my $coordset_A_aref (@$coordsets_A_aref) { + + foreach my $coordset_B_aref (@$coordsets_B_aref) { + + $sum += &overlap_length($coordset_A_aref, $coordset_B_aref); + } + } + + return($sum); +} + + +## overlap, compatible, and A contains B +sub compatible_overlap_A_contains_B { + my ($coordsets_A_aref, $coordsets_B_aref) = @_; + + my ($A_lend, $A_rend) = &get_coordset_span(@$coordsets_A_aref); + my ($B_lend, $B_rend) = &get_coordset_span(@$coordsets_B_aref); + + + if (&contains([$A_lend, $A_rend], [$B_lend, $B_rend]) + && + &compatible_overlap($coordsets_A_aref, $coordsets_B_aref) ) { + + return(1); + } + else { + return(0); + } +} + + + +## overlap and encode identical internal boundaries. +sub compatible_overlap { + my ($coordsets_A_aref, $coordsets_B_aref) = @_; + + my $verbose_check = 0; + + print "* Compatibility check between: " . Dumper($coordsets_A_aref) . " and " . Dumper($coordsets_B_aref) if $verbose_check; + + + my @A_coordsets = &_order_coordsets(@$coordsets_A_aref); + my @B_coordsets = &_order_coordsets(@$coordsets_B_aref); + + my @A_span = &get_coordset_span(@A_coordsets); + my @B_span = &get_coordset_span(@B_coordsets); + + unless (&overlap(\@A_span, \@B_span)) { + print "-no overlap of spans: @A_span, @B_span\n" if $verbose_check; + return(0); + } + + + my ($i, $j); + my $found_overlap_flag = 0; + + overlap_search: + for ($i = 0; $i <= $#A_coordsets; $i++) { + + for ($j = 0; $j <= $#B_coordsets; $j++) { + + if (&overlap($A_coordsets[$i], $B_coordsets[$j])) { + + + $found_overlap_flag = 1; + + + + last overlap_search; + } + } + } + + unless ($found_overlap_flag) { + print "-no overlap of segments.\n" if $verbose_check; + return(0); + } + + ## if its not the first segment of A or B, then they're incompatible. + if (! ($i == 0 || $j == 0) ) { + print "-overlap doesn't anchor at first segment of either entry. $i, $j\n" if $verbose_check; + return(0); + } + + + while ($i <= $#A_coordsets && $j <= $#B_coordsets) { + + my @a_coords = @{$A_coordsets[$i]}; + my @b_coords = @{$B_coordsets[$j]}; + + + ## left junction check. + if ($i != 0 && $j != 0) { + ## left bounds should have matching edges. + if ($a_coords[0] != $b_coords[0]) { + print "-(internal seg) left bounds fail to match: " . Dumper(\@A_coordsets) . Dumper(\@B_coordsets) if $verbose_check; + return(0); + } + } + ## right junction check: + if ($i != $#A_coordsets && $j != $#B_coordsets) { + if ($a_coords[1] != $b_coords[1]) { + print "-(internal seg) right bounds fail to match: " . Dumper(\@A_coordsets) . Dumper(\@B_coordsets) if $verbose_check; + return(0); + } + } + + ## left bound check + if ($i == 0 && $j != 0) { + if ($a_coords[0] < $b_coords[0]) { + print "-(i first seg), fail left: " . Dumper(\@A_coordsets) . Dumper(\@B_coordsets) if $verbose_check; + return(0); + } + } + if ($i != 0 && $j == 0) { + if ($b_coords[0] < $a_coords[0]) { + print "-(j first seg), fail left: " . Dumper(\@A_coordsets) . Dumper(\@B_coordsets) if $verbose_check; + return(0); + } + } + + ## right bound check. + if ($i == $#A_coordsets && $j != $#B_coordsets) { + if ($a_coords[1] > $b_coords[1]) { + print "-(i last seg), fail right: " . Dumper(\@A_coordsets) . Dumper(\@B_coordsets) if $verbose_check; + return(0); + } + } + + if ($i != $#A_coordsets && $j == $#B_coordsets) { + if ($b_coords[1] > $a_coords[1]) { + print "-(j laset seg), fail right: " . Dumper(\@A_coordsets) . Dumper(\@B_coordsets) if $verbose_check; + return(0); + } + } + + $i++; + $j++; + + } + + ## must be compatible + print "-made it. Must be compatible.\n" if $verbose_check; + return(1); + + + +} + +#### +sub _order_coordsets { + my (@coordsets) = @_; + + my @ret_coords; + + foreach my $coordpair (@coordsets) { + my ($lend, $rend) = sort {$a<=>$b} @$coordpair; + push (@ret_coords, [$lend, $rend]); + } + + + @ret_coords = sort {$a->[0]<=>$b->[0]} @ret_coords; + + return(@ret_coords); +} + +sub get_gap_coords { + my ($coordsets_aref) = @_; + + my @coordsets = &_order_coordsets(@$coordsets_aref); + + my @gaps; + + for (my $i = 1; $i <= $#coordsets; $i++) { + + my $prev_rend = $coordsets[$i-1]->[1]; + my $curr_lend = $coordsets[$i]->[0]; + + push (@gaps, [$prev_rend + 1, $curr_lend - 1]); + } + + return(@gaps); +} + + + +sub get_coordset_span { + my @coordsets = @_; + + my @coords; + foreach my $coordset (@coordsets) { + push (@coords, @$coordset); + } + + my $min_coord = min(@coords); + my $max_coord = max(@coords); + + return($min_coord, $max_coord); +} + + +sub coordsets_to_string { + my @coordsets = @_; + + my @text; + foreach my $coordset (@coordsets) { + my ($lend, $rend) = @$coordset; + push (@text, "($lend,$rend)"); + } + + my $text_line = join("--", @text); + return($text_line); +} + + + + +sub _coords_are_identical { + my ($coordpair_A_aref, $coordpair_B_aref) = @_; + + if ($coordpair_A_aref->[0] == $coordpair_B_aref->[0] + && + $coordpair_A_aref->[1] == $coordpair_B_aref->[1]) { + + return(1); + } + + else { + return(0); + } +} + + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/Overlap_piler.pm b/99.scripts/trinity_utils/PerlLib/Overlap_piler.pm new file mode 100644 index 0000000..23ac46e --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Overlap_piler.pm @@ -0,0 +1,174 @@ +package main; +our $SEE = 0; + +## taken from CDNA::Overlap_assembler.pm +## really should be a more general purpose class as written here. + + +package Overlap_piler; + + +use strict; +use warnings; +use Carp; + +sub new { + my $packagename = shift; + my $self = { + node_list => [] + }; + bless ($self, $packagename); + return ($self); +} + + +#### Static method!!! +sub simple_coordsets_collapser { + my @coordsets = @_; ## list of coordinates [a, b], [c, d], ... + + my $counter = 0; + + my $piler = new Overlap_piler(); + + my %coords_mapping; + foreach my $coordset (@coordsets) { + $counter++; + $coords_mapping{$counter} = [@$coordset]; + + my ($lend, $rend) = @$coordset; + if ($lend !~ /\d/ || $rend !~ /\d/) { + confess "Error, coordinates [ $lend, $rend ] include a non-number"; + } + + $piler->add_coordSet($counter, @$coordset); + } + + my @clusters = $piler->build_clusters(); + + my @coord_spans; + foreach my $cluster (@clusters) { + my @eles = @$cluster; + my @coords; + foreach my $ele (@eles) { + push (@coords, @{$coords_mapping{$ele}}); + } + + @coords = sort {$a<=>$b} @coords; + + my $min_coord = shift @coords; + my $max_coord = pop @coords; + + push (@coord_spans, [$min_coord, $max_coord]); + } + + return (@coord_spans); + +} + + + +#### +sub add_coordSet { + my $self = shift; + my ($acc, $end5, $end3) = @_; + my ($lend, $rend) = sort {$a<=>$b} ($end5, $end3); + my $node = CoordSet_node->new($acc, $lend, $rend); + push (@{$self->{node_list}}, $node); +} + + + + +#### +sub build_clusters { + my $self = shift; + my $node_list_aref = $self->{node_list}; + @{$node_list_aref} = sort {$a->{lend}<=>$b->{lend}} @{$node_list_aref}; #sort by lend coord. + ## set indices + for (my $i = 0; $i <= $#{$node_list_aref}; $i++) { + $node_list_aref->[$i]->{myIndex} = $i; + } + + my @clusters; + my $first_node = $node_list_aref->[0]; + my $start_pos = 0; + my ($exp_left, $exp_right) = ($first_node->{lend}, $first_node->{rend}); + print $first_node->{acc} . " ($exp_left, $exp_right)\n" if $SEE; + for (my $i = 1; $i <= $#{$node_list_aref}; $i++) { + my $curr_node = $node_list_aref->[$i]; + my ($lend, $rend) = ($curr_node->{lend}, $curr_node->{rend}); + print $curr_node->{acc} . " ($lend, $rend)\n" if $SEE; + if ($exp_left <= $rend && $exp_right >= $lend) { #overlap + $exp_left = &min($exp_left, $lend); + $exp_right = &max($exp_right, $rend); + print "overlap. New expanded coords: ($exp_left, $exp_right)\n" if $SEE; + } else { + print "No overlap; Creating cluster: " if $SEE; + my @cluster; + for (my $j=$start_pos; $j < $i; $j++) { + my $acc = $node_list_aref->[$j]->{acc}; + push (@cluster, $acc); + print "$acc, " if $SEE; + } + push (@clusters, [@cluster]); + $start_pos = $i; + ($exp_left, $exp_right) = ($lend, $rend); + print "\nResetting expanded coords: ($lend, $rend)\n" if $SEE; + } + } + + print "# Adding final cluster.\n" if $SEE; + if ($start_pos != $#{$node_list_aref}) { + print "final cluster: " if $SEE; + my @cluster; + for (my $j = $start_pos; $j <= $#{$node_list_aref}; $j++) { + my $acc = $node_list_aref->[$j]->{acc}; + print "$acc, " if $SEE; + push (@cluster, $acc); + } + push (@clusters, [@cluster]); + print "\n" if $SEE; + } else { + my $acc = $node_list_aref->[$start_pos]->{acc}; + push (@clusters, [$acc]); + print "adding final $acc.\n" if $SEE; + } + return (@clusters); +} + +sub min { + my (@x) = @_; + @x = sort {$a<=>$b} @x; + my $min = shift @x; + return ($min); +} + +sub max { + my @x = @_; + @x = sort {$a<=>$b} @x; + my $max = pop @x; + return ($max); +} + +################################################################# +package CoordSet_node; +use strict; + +sub new { + my $packagename = shift; + my ($acc, $lend, $rend) = @_; + my $self = { acc=>$acc, + lend=>$lend, + rend=>$rend, + myIndex=>undef(), + overlapping_indices=>[] + }; + bless ($self, $packagename); + return ($self); +} + +1; #EOM + + + + diff --git a/99.scripts/trinity_utils/PerlLib/PSL_parser.pm b/99.scripts/trinity_utils/PerlLib/PSL_parser.pm new file mode 100644 index 0000000..801cfa6 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/PSL_parser.pm @@ -0,0 +1,297 @@ +package PSL_parser; + +use strict; +use warnings; + +use Carp; + +sub new { + my $packagename = shift; + + my ($psl_file) = @_; + + my $fh; + + if (ref $psl_file eq "IO::Handle") { + $fh = $psl_file; # a filehandle not a file + } + else { + + + unless (-e $psl_file) { + confess "error, cannot find $psl_file"; + } + open ($fh, $psl_file) or confess "Error, cannot open file $psl_file"; + } + + + my $self = { file => $psl_file, + fh => $fh, + }; + + bless ($self, $packagename); + + + + return($self); +} + +sub get_next { + my $self = shift; + + my $fh = $self->{fh}; + + while (my $line = <$fh>) { + if ($line =~ /^\d+\s/) { + return(PSL_entry->new($line)); + } + } + + return(undef); # no more lines +} + + + +################################################################################## +package PSL_entry; + +use strict; +use warnings; +use Carp; + + +sub new { + my $packagename = shift; + my ($psl_line) = @_; + + my @fields = split(/\t/, $psl_line); + + my $self = { fields => [@fields] }; + + bless ($self, $packagename); + + return($self); +} + + + +################ +# blat format: # Q=cDNA T=genomic +################ + +# 0: match +# 1: mis-match +# 2: rep. match +# 3: N's +# 4: Q gap count +# 5: Q gap bases +# 6: T gap count +# 7: T gap bases +# 8: strand +# 9: Q name +# 10: Q size +# 11: Q start +# 12: Q end +# 13: T name +# 14: T size +# 15: T start +# 16: T end +# 17: block count +# 18: block Sizes +# 19: Q starts +# 20: T starts +# 21: Q seqs (pslx format) +# 22: T seqs (pslx format) + +## All sequences start at 0 here; array-based. + + +sub get_line { + my $self = shift; + return(join("\t", @{$self->{fields}})); +} + + +sub get_match_count { + my $self = shift; + return($self->{fields}->[0]); +} + +sub get_mismatch_count { + my $self = shift; + return($self->{fields}->[1]); +} + +sub get_N_count { + my $self = shift; + return($self->{fields}->[3]); +} + +sub get_Q_gap_count { + my $self = shift; + return($self->{fields}->[4]); +} + +sub get_Q_gap_bases { + my $self = shift; + return($self->{fields}->[5]); +} + +sub get_T_gap_count { + my $self = shift; + return($self->{fields}->[6]); +} + +sub get_T_gap_bases { + my $self = shift; + return($self->{fields}->[7]); +} + +sub get_strand { + my $self = shift; + return($self->{fields}->[8]); +} + +sub get_Q_name { + my $self = shift; + return($self->{fields}->[9]); +} + +sub get_Q_size { + my $self = shift; + return($self->{fields}->[10]); +} + +sub get_Q_span { + my $self = shift; + return($self->{fields}->[11] + 1, $self->{fields}->[12]); +} + +sub get_T_name { + my $self = shift; + return($self->{fields}->[13]); +} + +sub get_T_size { + my $self = shift; + return($self->{fields}->[14]); +} + +sub get_T_span { + my $self = shift; + return($self->{fields}->[15] + 1, $self->{fields}->[16]); +} + + +#### +sub get_per_id { + my $self = shift; + + my $matches = $self->get_match_count(); + my $mismatches = $self->get_mismatch_count(); + + my $per_id = $matches / ($matches + $mismatches) * 100; + + $per_id = sprintf("%.2f", $per_id); + + return($per_id); +} + + + + +#### +sub get_alignment_coords { + my $self = shift; + + my @x = @{$self->{fields}}; + + + my @alignment_segments; + + + my @cdna_coords = split (/,/, $x[19]); + my @genomic_coords = split (/,/, $x[20]); + my @lengths = split (/,/, $x[18]); + + my $strand = $self->get_strand(); + my $cdna_length = $self->get_Q_size(); + + my @ret_genome_coords; + my @ret_cdna_coords; + + ## report each segment match as a separate btab entry: + my $segment_number = 0; + my $num_segs = scalar(@genomic_coords); + for (my $i = 0; $i < $num_segs; $i++) { + $segment_number++; + my $length = $lengths[$i]; + unless (defined $length) { next; } + my $cdna_coord = $cdna_coords[$i]; + my $genomic_coord = $genomic_coords[$i]; + my ($cdna_end5, $cdna_end3) = (++$cdna_coord, $cdna_coord + $length - 1); + my ($genomic_end5, $genomic_end3) = (++$genomic_coord, $genomic_coord + $length -1); + if ($strand eq "-") { + ($cdna_end5, $cdna_end3) = ($cdna_length - $cdna_end5 + 1, $cdna_length - $cdna_end3 + 1); + } + + push (@ret_genome_coords, [$genomic_end5, $genomic_end3]); + push (@ret_cdna_coords, [$cdna_end5, $cdna_end3]); + } + + + return(\@ret_genome_coords, \@ret_cdna_coords); +} + + +sub toString { + my $self = shift; + + my $genome_acc = $self->get_T_name(); + my $cdna_acc = $self->get_Q_name(); + + my $genome_length = $self->get_T_size(); + my $cdna_length = $self->get_Q_size(); + + + my $strand = $self->get_strand(); + + my ($genome_lend, $genome_rend) = $self->get_T_span(); + my ($cdna_lend, $cdna_rend) = $self->get_Q_span(); + + my $per_id = sprintf("%.2f", $self->get_per_id()); + + my $ret_text = join("\t", $genome_acc, "($genome_length)", "$genome_lend-$genome_rend", $strand, + $cdna_acc, "($cdna_length)", "$cdna_lend-$cdna_rend", $per_id); + + + my ($genome_coords_aref, $cdna_coords_aref) = $self->get_alignment_coords(); + + my @genome_coords = @$genome_coords_aref; + my @cdna_coords = @$cdna_coords_aref; + + my $align_text = ""; + while (@genome_coords) { + my $genome_coordset = shift @genome_coords; + my $cdna_coordset = shift @cdna_coords; + + if ($align_text) { + $align_text .= "...."; + } + $align_text .= $genome_coordset->[0] . "(" . $cdna_coordset->[0] . ")-" + . $genome_coordset->[1] . "(" . $cdna_coordset->[1] . ")"; + } + + $ret_text .= "\n$align_text\n"; + + return ($ret_text); +} + + +1; #EOM + + + + + + diff --git a/99.scripts/trinity_utils/PerlLib/Pipeliner.pm b/99.scripts/trinity_utils/PerlLib/Pipeliner.pm new file mode 100644 index 0000000..15e92be --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Pipeliner.pm @@ -0,0 +1,265 @@ +package Pipeliner; + +use strict; +use warnings; +use Carp; +use Cwd; + +################################ +## Verbose levels: +## 1: see CMD string +## 2: see stderr during process +################################ + + +#################### +## Static methods: +#################### + +#### +sub ensure_full_path { + my ($path, $ADD_GZ_FIFO_FLAG) = @_; + + unless ($path =~ m|^/|) { + $path = cwd() . "/$path"; + } + + if ($ADD_GZ_FIFO_FLAG && $path =~ /\.gz$/) { + $path = "<(zcat $path)"; + } + + return($path); +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, CMD: $cmd died with ret $ret"; + } + return; +} + + +################ +## Obj methods: +################ + +#### +sub new { + my $packagename = shift; + my %params = @_; + + my $VERBOSE = 0; + if ($params{-verbose}) { + $VERBOSE = $params{-verbose}; + } + my $cmds_log = $params{-cmds_log}; + + + my $self = { + cmd_objs => [], + checkpoint_dir => undef, + cmds_log_ofh => undef, + VERBOSE => $VERBOSE, + }; + + bless ($self, $packagename); + + if (my $checkpoint_dir = $params{-checkpoint_dir}) { + $self->set_checkpoint_dir($checkpoint_dir); + } + + unless ($cmds_log) { + $cmds_log = "pipeliner.$$.cmds"; + + if (my $checkpoint_dir = $self->get_checkpoint_dir) { + $cmds_log = "$checkpoint_dir/$cmds_log"; + } + } + + # open cmds log + open (my $ofh, ">$cmds_log") or confess "Error, cannot write to $cmds_log"; + $self->{cmds_log_ofh} = $ofh; + + + return($self); +} + + +sub add_commands { + my $self = shift; + my @cmds = @_; + + foreach my $cmd (@cmds) { + unless (ref($cmd) =~ /Command/) { + confess "Error, need Command object as param"; + } + + my $checkpoint_file = $cmd->get_checkpoint_file(); + if ($checkpoint_file !~ m|^/|) { + if (my $checkpoint_dir = $self->get_checkpoint_dir()) { + $checkpoint_file = "$checkpoint_dir/$checkpoint_file"; + $cmd->reset_checkpoint_file($checkpoint_file); + } + } + + + push (@{$self->{cmd_objs}}, $cmd); + } + + return $self; + +} + +sub set_checkpoint_dir { + my $self = shift; + my ($checkpoint_dir) = @_; + $checkpoint_dir = &ensure_full_path($checkpoint_dir); + if (! -d $checkpoint_dir) { + mkdir($checkpoint_dir) or die "Error, cannot mkdir $checkpoint_dir"; + } + $self->{checkpoint_dir} = $checkpoint_dir; +} + +sub get_checkpoint_dir { + my $self = shift; + return($self->{checkpoint_dir}); +} + +sub has_commands { + my $self = shift; + if ($self->_get_commands()) { + return(1); + } + else { + return(0); + } +} + +sub run { + my $self = shift; + my $VERBOSE = $self->{VERBOSE}; + + my $cmds_log_ofh = $self->{cmds_log_ofh}; + + foreach my $cmd_obj ($self->_get_commands()) { + + my $cmdstr = $cmd_obj->get_cmdstr(); + print $cmds_log_ofh "$cmdstr\n"; + + my $msg = $cmd_obj->{msg}; + + my $checkpoint_file = $cmd_obj->get_checkpoint_file(); + + if (-e $checkpoint_file) { + print STDERR "-- Skipping CMD: $cmdstr, checkpoint [$checkpoint_file] exists.\n" if $VERBOSE; + } + else { + my $datestamp = localtime(); + print STDERR "* [$datestamp] Running CMD: $cmdstr\n" if $VERBOSE; + + my $tmp_stderr = "tmp.$$." . time() . ".stderr"; + if (-e $tmp_stderr) { + unlink($tmp_stderr); + } + + if ($VERBOSE < 2 && $cmdstr !~ / 2>/ ) { + $cmdstr .= " 2>$tmp_stderr"; + } + + print STDERR $msg if $msg; + + my $ret = system($cmdstr); + if ($ret) { + + if (-e $tmp_stderr) { + my $errmsg = `cat $tmp_stderr`; + if ($errmsg =~ /\w/) { + print STDERR "\n\nError encountered:: \n\n"; + } + unlink($tmp_stderr); + } + + confess "Error, cmd: $cmdstr died with ret $ret $!"; + } + else { + `touch $checkpoint_file`; + if ($?) { + + confess "Error creating checkpoint file: $checkpoint_file"; + } + } + + if (-e $tmp_stderr) { + unlink($tmp_stderr); + } + } + } + + + # reset in case reusing the pipeline obj + $self->{cmd_objs} = []; # reinit + + + return; +} + +sub _get_commands { + my $self = shift; + + return(@{$self->{cmd_objs}}); +} + + + + + +package Command; +use strict; +use warnings; +use Carp; + +sub new { + my $packagename = shift; + + my ($cmdstr, $checkpoint_file, $message) = @_; + + unless ($cmdstr && $checkpoint_file) { + confess "Error, need cmdstr and checkpoint filename as params"; + } + + my $self = { cmdstr => $cmdstr, + checkpoint_file => $checkpoint_file, + msg => $message, + }; + + bless ($self, $packagename); + + return($self); +} + +#### +sub get_cmdstr { + my $self = shift; + return($self->{cmdstr}); +} + +#### +sub get_checkpoint_file { + my $self = shift; + return($self->{checkpoint_file}); +} + +#### +sub reset_checkpoint_file { + my $self = shift; + my $checkpoint_file = shift; + + $self->{checkpoint_file} = $checkpoint_file; +} + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/Process_cmd.pm b/99.scripts/trinity_utils/PerlLib/Process_cmd.pm new file mode 100644 index 0000000..a9d724f --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Process_cmd.pm @@ -0,0 +1,55 @@ +package Process_cmd; + +use strict; +use warnings; +use Carp; +use Cwd; + +require Exporter; +our @ISA = qw(Exporter); +our @EXPORT = qw(process_cmd ensure_full_path); + + +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + if ($ret) { + confess "Error, cmd:\n$cmd\n died with ret ($ret)"; + } + + return; +} + + +sub ensure_full_path { + my ($path) = @_; + + my @ret_paths; + + foreach my $p (split(/,\s*/, $path)) { + push (@ret_paths, &ensure_full_single_path($p)); + } + + my $ret_path = join(",", @ret_paths); + + return($ret_path); + +} + +sub ensure_full_single_path { + my ($path) = @_; + + unless ($path =~ m|^/|) { + $path = cwd() . "/$path"; + } + + return($path); +} + + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/SAM_entry.pm b/99.scripts/trinity_utils/PerlLib/SAM_entry.pm new file mode 100644 index 0000000..b6f1655 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/SAM_entry.pm @@ -0,0 +1,629 @@ +package SAM_entry; + +use strict; +use warnings; +use Carp; + + + +sub new { + my $packagename = shift; + my ($line) = @_; + + unless (defined $line) { + confess "Error, need sam text line as parameter"; + } + + chomp $line; + + my @fields = split(/\t/, $line); + + my $self = { + _line => $line, + _fields => [@fields], + }; + + bless ($self, $packagename); + + return($self); +} + + +#### +sub get_original_line { + my $self = shift; + return($self->{_line}); +} + +#### +sub get_fields { + my $self = shift; + return (@{$self->{_fields}}); +} + + +#### +sub get_read_name { + my $self = shift; + return ($self->{_fields}->[0]); +} + +#### +sub reconstruct_full_read_name { + my $self = shift; + + my $read_name = $self->get_core_read_name(); + + if ($self->is_first_in_pair()) { + $read_name .= "/1"; + } + elsif ($self->is_second_in_pair()) { + $read_name .= "/2"; + } + + return($read_name); +} + + +#### +sub get_core_read_name { + my $self = shift; + my $read_name = $self->get_read_name(); + my $core_read_name = $read_name; + + $core_read_name =~ s|/\d$||; + + return($core_read_name); +} + + + +#### +sub get_scaffold_name { + my $self = shift; + return($self->{_fields}->[2]); +} + +#### +sub get_aligned_position { + my $self = shift; + return($self->{_fields}->[3]); +} + +sub get_scaffold_position { # preferred + my $self = shift; + return($self->get_aligned_position()); +} + + +sub get_scaffold_start_position { + my $self = shift; + my ($lend, $rend) = $self->get_genome_span(); + + my $strand = $self->get_query_strand(); + if ($strand eq '+') { + return($lend); + } + else { + return($rend); + } +} + + +#### +sub get_read_group { + my $self = shift; + my $line = $self->get_original_line(); + + if ($line =~ /RG:Z:(\S+)/) { + return($1); + } + else { + return undef; + } +} + + +#### +sub get_cigar_alignment { + my $self = shift; + return($self->{_fields}->[5]); +} + +### +sub get_genome_span { + my $self = shift; + my ($genome_aref, $read_aref) = $self->get_alignment_coords(); + + my @coords; + foreach my $genome_coordset (@$genome_aref) { + push (@coords, @$genome_coordset); + } + + @coords = sort {$a<=>$b} @coords; + + my $min_coord = shift @coords; + my $max_coord = pop @coords; + + return($min_coord, $max_coord); +} + +#### +sub get_read_span { + my $self = shift; + + my ($genome_aref, $read_aref) = $self->get_alignment_coords(); + + my @coords; + foreach my $read_coordset (@$read_aref) { + push (@coords, @$read_coordset); + } + + @coords = sort {$a<=>$b} @coords; + + my $min_coord = shift @coords; + my $max_coord = pop @coords; + + return($min_coord, $max_coord); + +} + +#### +sub get_alignment_length { + my $self = shift; + + my ($genome_coords_aref, $read_coords_aref) = $self->get_alignment_coords(); + + my $sum_len = 0; + + my @genome_coords = @$genome_coords_aref; + foreach my $coords (@genome_coords) { + my ($genome_lend, $genome_rend) = @$coords; + + $sum_len += abs($genome_rend - $genome_lend) + 1; + } + + return($sum_len); +} + + +#### +sub get_alignment_coords { + my $self = shift; + + my $genome_lend = $self->get_aligned_position(); + + my $alignment = $self->get_cigar_alignment(); + + my $query_lend = 0; + + my @genome_coords; + my @query_coords; + + + my $sum_hardmasked_query = 0; + + $genome_lend--; # move pointer just before first position. + + while ($alignment =~ /(\d+)([A-Z])/g) { + my $len = $1; + my $code = $2; + + unless ($code =~ /^[MSDNIH]$/) { + confess "Error, cannot parse cigar code [$code] " . $self->toString(); + } + + # print "parsed $len,$code\n"; + + if ($code eq 'M') { # aligned bases match or mismatch + + my $genome_rend = $genome_lend + $len; + my $query_rend = $query_lend + $len; + + push (@genome_coords, [$genome_lend+1, $genome_rend]); + push (@query_coords, [$query_lend+1, $query_rend]); + + # reset coord pointers + $genome_lend = $genome_rend; + $query_lend = $query_rend; + + } + elsif ($code eq 'D' || $code eq 'N') { # insertion in the genome or gap in query (intron, perhaps) + $genome_lend += $len; + + } + + elsif ($code eq 'I' # gap in genome or insertion in query + || + $code eq 'S' || $code eq 'H') # masked region of query + { + $query_lend += $len; + + if ($code eq "H") { + $sum_hardmasked_query += $len; + } + } + } + + + ## see if reverse strand alignment - if so, must revcomp the read matching coordinates. + if ($self->get_query_strand() eq '-') { + + my $read_len = length($self->get_sequence()); + unless ($read_len) { + confess "Error, no read length obtained from entry: " . $self->get_original_line(); + } + $read_len += $sum_hardmasked_query; + + my @revcomp_coords; + foreach my $coordset (@query_coords) { + my ($lend, $rend) = @$coordset; + + my $new_lend = $read_len - $lend + 1; + my $new_rend = $read_len - $rend + 1; + + push (@revcomp_coords, [$new_lend, $new_rend]); + } + + @query_coords = @revcomp_coords; + + } + + + + return(\@genome_coords, \@query_coords); +} + + +#### +sub get_mate_scaffold_name { + my $self = shift; + + return($self->{_fields}->[6]); +} + + +#### +sub set_mate_scaffold_name { + my $self = shift; + my $mate_scaffold_name = shift; + + $self->{_fields}->[6] = $mate_scaffold_name; + + return; +} + + +#### +sub get_mate_scaffold_position { + my $self = shift; + + return($self->{_fields}->[7]); +} + + +#### +sub set_mate_scaffold_position { + my $self = shift; + my $scaff_pos = shift; + + $self->{_fields}->[7] = $scaff_pos; + + return; +} + + +#### +sub toString { + my $self = shift; + my @fields = @{$self->{_fields}}; + + if ($self->is_paired()) { + $fields[0] = $self->get_core_read_name(); + } + + return( join("\t", @fields)); +} + + +#### +sub get_mapping_quality { + my $self = shift; + return($self->{_fields}->[4]); +} + + +#### +sub get_sequence { + my $self = shift; + return($self->{_fields}->[9]); +} + +#### +sub get_quality_scores { + my $self = shift; + return($self->{_fields}->[10]); +} + + +#### +sub get_inferred_insert_size { + my $self = shift; + + # should probably check to see if it's a paired read or not... user beware + return($self->{_fields}->[8]); +} + + + +################### +## Flag Processing +################### + +# from sam format spec: + +=flag_description + +Flag Description +0x0001 the read is paired in sequencing, no matter whether it is mapped in a pair +0x0002 the read is mapped in a proper pair (depends on the protocol, normally inferred during alignment) 1 +0x0004 the query sequence itself is unmapped +0x0008 the mate is unmapped 1 +0x0010 strand of the query (0 for forward; 1 for reverse strand) +0x0020 strand of the mate 1 +0x0040 the read is the first read in a pair 1,2 +0x0080 the read is the second read in a pair 1,2 +0x0100 the alignment is not primary (a read having split hits may have multiple primary alignment records) +0x0200 the read fails platform/vendor quality checks +0x0400 the read is either a PCR duplicate or an optical duplicate + +1. Flag 0x02, 0x08, 0x20, 0x40 and 0x80 are only meaningful when flag 0x01 is present. +2. If in a read pair the information on which read is the first in the pair is lost in the upstream analysis, flag 0x01 shuld +be present and 0x40 and 0x80 are both zero. + +=cut + + +#### +sub get_flag { + my $self = shift; + my $flag = $self->{_fields}->[1]; + return($flag); +} + +sub set_flag { + my $self = shift; + my $flag = shift; + + unless (defined $flag) { + confess "Error, need flag value"; + } + + $self->{_fields}->[1] = $flag; + return; +} + +#### +sub is_paired { + my $self = shift; + return($self->_get_bit_val(0x0001)); +} + +sub set_paired { + my $self = shift; + my $bit_val = shift; + + $self->_set_bit_val(0x0001, $bit_val); + + return; +} + +#### +sub is_proper_pair { + my $self = shift; + return($self->_get_bit_val(0x0002)); +} + +sub set_proper_pair { + my $self = shift; + my $bit_val = shift; + + $self->_set_bit_val(0x0002, $bit_val); + return; +} + +#### +sub is_query_unmapped { + my $self = shift; + return($self->_get_bit_val(0x0004)); +} + +sub set_query_unmapped { + my $self = shift; + my $bit_val = shift; + + $self->_set_bit_val(0x0004, $bit_val); +} + + +#### +sub is_mate_unmapped { + my $self = shift; + return($self->_get_bit_val(0x0008)); +} + +sub set_mate_unmapped { + my $self = shift; + my $bit_val = shift; + + return($self->_set_bit_val(0x0008, $bit_val)); +} + +sub is_duplicate { + my ($self) = shift; + + return($self->_get_bit_val(0x0400)); +} +sub set_duplicate { + my $self = shift; + my $bit_val = shift; + + return($self->_set_bit_val(0x0400, $bit_val)); +} + + +#### +sub get_query_strand { + my $self = shift; + + my $strand = ($self->_get_bit_val(0x0010)) ? '-' : '+'; + return($strand); +} + +#### +sub get_query_transcribed_strand { + my $self = shift; + my ($SS_lib_type) = @_; + + unless ($SS_lib_type) { + confess "Error, SS_lib_type required as a parameter, possible values: RF,FR,F,R " . $self->toString(); + } + + + my $aligned_strand = $self->get_query_strand(); + my $opposite_strand = ($aligned_strand eq '+') ? '-' : '+'; + + my $transcribed_strand; + + if (! $self->is_paired()) { + + ## UNPAIRED or SINGLE READS + unless ($SS_lib_type =~ /^(F|R)$/) { + confess "Error, cannot have $SS_lib_type library type with unpaired reads " . $self->toString(); + } + + $transcribed_strand = ($SS_lib_type eq "F") ? $aligned_strand : $opposite_strand; + } + else { + + ## paired RNA-Seq reads: left fragment is on the 3' end revcomped, and right fragment is at the 5' end sense strand. + + unless ($SS_lib_type =~ /^(FR|RF)$/) { + confess "Error, cannot have $SS_lib_type library type with paired reads " . $self->toString(); + } + + if ($self->is_first_in_pair()) { + $transcribed_strand = ($SS_lib_type eq "FR") ? $aligned_strand : $opposite_strand; + } + else { + # second pair + $transcribed_strand = ($SS_lib_type eq "FR") ? $opposite_strand : $aligned_strand; + } + } + + + return($transcribed_strand); + +} + + + + +sub set_query_strand { + my $self = shift; + my $strand = shift; + + unless ($strand eq '+' || $strand eq '-') { + confess "Error, strand value must be [+-]"; + } + + my $bit_val = ($strand eq '+') ? 0 : 1; + $self->_set_bit_val(0x0010, $bit_val); +} + +#### +sub get_mate_strand { + my $self = shift; + + my $strand = ($self->_get_bit_val(0x0020)) ? '-' : '+'; + return($strand); +} + +sub set_mate_strand { + my $self = shift; + my $strand = shift; + + unless ($strand eq '+' || $strand eq '-') { + confess "Error, strand value must be [+-]"; + } + + my $bit_val = ($strand eq '+') ? 0 : 1; + $self->_set_bit_val(0x0020, $bit_val); +} + +#### +sub is_first_in_pair { + my $self = shift; + return($self->_get_bit_val(0x0040)); +} + +sub set_first_in_pair { + my $self = shift; + my $bit_val = shift; + + $self->_set_bit_val(0x0040, $bit_val); + return; +} + +#### +sub is_second_in_pair { + my $self = shift; + return($self->_get_bit_val(0x0080)); +} + + +sub set_second_in_pair { + my $self = shift; + my $bit_val = shift; + + $self->_set_bit_val(0x0080, $bit_val); + return; +} + + + +#### +sub _get_bit_val { + my $self = shift; + my ($bit_position) = @_; + + my $flag = $self->get_flag(); + return($flag & $bit_position); +} + + +#### +sub _set_bit_val { + my $self = shift; + my ($bit_position, $bit_val) = @_; + + unless (defined $bit_position && defined $bit_val) { + confess "Error, need bit position and value"; + } + + my $flag = $self->get_flag(); + + if ($bit_val) { + $flag |= $bit_position; + } + else { + # erase bit + $flag &= ~$bit_position; + } + + $self->set_flag($flag); +} + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/SAM_reader.pm b/99.scripts/trinity_utils/PerlLib/SAM_reader.pm new file mode 100644 index 0000000..d61c9b2 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/SAM_reader.pm @@ -0,0 +1,109 @@ +package SAM_reader; + +use strict; +use warnings; +use Carp; + +use SAM_entry; + +sub new { + my $packagename = shift; + my $filename = shift; + + unless ($filename) { + confess "Error, need SAM filename as parameter"; + } + + my $self = { filename => $filename, + _next => undef, + _fh => undef, + }; + + bless ($self, $packagename); + + $self->_init(); + + return($self); +} + + +#### +sub _init { + my ($self) = @_; + + if ($self->{filename} =~ /\.bam$/) { + open ($self->{_fh}, "samtools view $self->{filename} |") or confess "Error, cannot open file " . $self->{filename}; + } + elsif ($self->{filename} =~ /\.gz$/) { + open ($self->{_fh}, "gunzip -c $self->{filename} | ") or confess "Error, cannot open file " . $self->{filename}; + } + else { + open ($self->{_fh}, $self->{filename}) or confess "Error, cannot open file " . $self->{filename}; + } + + $self->_advance(); + + return; +} + +#### +sub _advance { + my ($self) = @_; + + my $fh = $self->{_fh}; + + my $next_line = <$fh>; + while (defined ($next_line) && ($next_line =~ /^\@/ || $next_line !~ /\w/)) { ## skip over sam headers + $next_line = <$fh>; + } + + if ($next_line) { + $self->{_next} = new SAM_entry($next_line); + } + else { + $self->{_next} = undef; + } + + return; +} + +#### +sub has_next { + my $self = shift; + + if (defined $self->{_next}) { + return(1); + } + else { + return(0); + } +} + + +#### +sub get_next { + my $self = shift; + + my $next_entry = $self->{_next}; + + $self->_advance(); + + if (defined $next_entry) { + return($next_entry); + } + else { + return(undef); + } +} + +#### +sub preview_next { + my $self = shift; + return($self->{_next}); +} + + +1; + + + diff --git a/99.scripts/trinity_utils/PerlLib/Simulate/Uniform_Read_Generator.pm b/99.scripts/trinity_utils/PerlLib/Simulate/Uniform_Read_Generator.pm new file mode 100644 index 0000000..f1abcbb --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Simulate/Uniform_Read_Generator.pm @@ -0,0 +1,327 @@ +package Simulate::Uniform_Read_Generator; + +use strict; +use warnings; +use Carp; +use Data::Dumper; + +sub new { + my $packagename = shift; + my $params_struct = shift; + + # params should have format: + # { + # coordsets => [ [lend,rend], [lend,rend], ...], + # mean_fragment_length => int, + # fragment_length_stdev => int, + # read_length => int, + # } + + unless (ref $params_struct eq "HASH") { + confess "Error, params struct required"; + } + + + my $coordsets_aref = $params_struct->{coordsets} or confess "Error, need coordsets parameter"; + my $mean_fragment_length = $params_struct->{mean_fragment_length} or confess "Error, need mean_fragment_length parameter"; + my $fragment_length_stdev = $params_struct->{fragment_length_stdev} or confess "Error, need fragment_length_stdev parameter"; + my $read_length = $params_struct->{read_length} or confess "Error, need read_length parameter"; + + my $self = { _coordsets => undef, # note, this is reconfigured below, not the same as the input parameter + mean_fragment_length => $mean_fragment_length, + fragment_length_stdev => $fragment_length_stdev, + read_length => $read_length, + }; + + bless($self, $packagename); + + $self->_init_coordsets($coordsets_aref); + + return($self); +} + + + +#### +sub _init_coordsets { + my $self = shift; + my ($coordsets_aref) = @_; + + my @ordered_coordsets = sort {$a->[0]<=>$b->[0]} @$coordsets_aref; + + my @coord_structs; + + my $prev_cdna_coord = 0; + foreach my $coordset (@ordered_coordsets) { + + my ($lend, $rend) = @$coordset; + + my $cdna_lend = $prev_cdna_coord + 1; ## cdna-relative coordinates + my $cdna_rend = $cdna_lend + ($rend - $lend); + + my $struct = { lend => $lend, + rend => $rend, + + cdna_lend => $cdna_lend, + cdna_rend => $cdna_rend, + }; + + push (@coord_structs, $struct); + + $prev_cdna_coord = $cdna_rend; + + } + + $self->{_coordsets} = \@coord_structs; + + return; +} + + +#### +sub simulate_paired_reads_random_pos { + my $self = shift; + my ($num_reads) = @_; + + unless (defined($num_reads) && $num_reads =~ /\d/) { + confess "Error, need num reads to simulate"; + } + + my $frag_length = $self->{mean_fragment_length}; + + my $max_cdna_rend = $self->_get_max_cdna_rend(); + + if ($frag_length > $max_cdna_rend) { + $frag_length = $max_cdna_rend; + } + + + my @reads; # to contain ( [ left_read_aref, right_read_aref], ... ) + + for (1..$num_reads) { + + my $read_start_pos = int( rand($max_cdna_rend - $frag_length + 1)) + 1; # uniform selection of potential start sites. + + # simulate a fragment according to fragment length + my @fragment = $self->_simulate_fragment($read_start_pos); + + # sample from each end to generate reads + my @left_read_segs = $self->_get_left_fragment_read(@fragment); + my @right_read_segs = $self->_get_right_fragment_read(@fragment); + + push (@reads, [ [@left_read_segs], [@right_read_segs] ] ); + + ## simulate reads from fragment + ## always do pairs and let the caller decide on which end or both to leverage. + + } + + + return(@reads); + +} + + +#### +sub simulate_paired_reads_uniformly_across_seq { + my $self = shift; + + my $frag_length = $self->{mean_fragment_length}; + + my $max_cdna_rend = $self->_get_max_cdna_rend(); + + if ($frag_length > $max_cdna_rend) { + $frag_length = $max_cdna_rend; + } + + my @reads; # to contain ( [ left_read_aref, right_read_aref], ... ) + + for my $read_start_pos (1..($max_cdna_rend - $frag_length + 1)) { + + # simulate a fragment according to fragment length + my @fragment = $self->_simulate_fragment($read_start_pos); + + # sample from each end to generate reads + my @left_read_segs = $self->_get_left_fragment_read(@fragment); + my @right_read_segs = $self->_get_right_fragment_read(@fragment); + + push (@reads, [ [@left_read_segs], [@right_read_segs] ] ); + + ## simulate reads from fragment + ## always do pairs and let the caller decide on which end or both to leverage. + + } + + + return(@reads); + +} + + + +#### +sub _simulate_fragment { + my $self = shift; + my ($read_start_pos) = @_; + + my $frag_length = $self->{mean_fragment_length}; + + my $max_cdna_rend = $self->_get_max_cdna_rend(); + + if ($frag_length > $max_cdna_rend) { + $frag_length = $max_cdna_rend; + } + + my @read_coords; + + my $structs_aref = $self->{_coordsets}; + + my $read_cdna_lend_pos = $read_start_pos; #$self->_get_cdna_coord_via_genome_coord($read_start_pos); + my $len_remaining = $frag_length; + + foreach my $struct (@$structs_aref) { + + my $exon_genome_lend = $struct->{lend}; + my $exon_genome_rend = $struct->{rend}; + + my $exon_cdna_lend = $struct->{cdna_lend}; + my $exon_cdna_rend = $struct->{cdna_rend}; + + + if ($exon_cdna_lend <= $read_cdna_lend_pos && $read_cdna_lend_pos <= $exon_cdna_rend) { + + ## convert read lend position to genome coordinate position. + my $delta = $read_cdna_lend_pos - $exon_cdna_lend; + my $read_genome_lend = $exon_genome_lend + $delta; + + my $read_exon_len = $exon_cdna_rend - $read_cdna_lend_pos + 1; + + if ($read_exon_len >= $len_remaining) { + my $read_genome_rend = $read_genome_lend + $len_remaining -1; + push (@read_coords, [$read_genome_lend, $read_genome_rend]); + last; + } + else { + my $read_genome_rend = $exon_genome_rend; + my $len_added = $read_genome_rend - $read_genome_lend + 1; + push (@read_coords, [$read_genome_lend, $read_genome_rend]); + $len_remaining -= $len_added; + $read_cdna_lend_pos += $len_added; + } + } + + + } + + return(@read_coords); + +} + + + + +#### +sub _get_max_cdna_rend { + my $self = shift; + + my $structs_aref = $self->{_coordsets}; + + my $max_cdna_rend = $structs_aref->[$#$structs_aref]->{cdna_rend}; + + return($max_cdna_rend); +} + + +#### +sub _get_cdna_coord_via_genome_coord { + my $self = shift; + my ($cdna_coord) = @_; + + my $structs_aref = $self->{_coordsets}; + + foreach my $struct (@$structs_aref) { + my $exon_genome_lend = $struct->{lend}; + my $exon_genome_rend = $struct->{rend}; + + my $exon_cdna_lend = $struct->{cdna_lend}; + my $exon_cdna_rend = $struct->{cdna_rend}; + + if ($cdna_coord >= $exon_cdna_lend && $cdna_coord <= $exon_cdna_rend) { + + my $delta = $cdna_coord - $exon_cdna_lend; + my $genome_coord = $exon_genome_lend += $delta; + return($genome_coord); + } + } + + confess "Error, could not map coordinate $cdna_coord within coordsets: " . Dumper($structs_aref); +} + + + +#### +sub _get_left_fragment_read { + my $self = shift; + my @frag = @_; + + my $read_length = $self->{read_length}; + + my $length_remaining = $read_length; + + my @read_coords; + + foreach my $segment (@frag) { + my ($frag_lend, $frag_rend) = @$segment; + my $seg_len = $frag_rend - $frag_lend + 1; + + if ($seg_len >= $length_remaining) { + my $read_seg = [$frag_lend, $frag_lend + $length_remaining - 1]; + push (@read_coords, $read_seg); + last; + } + else { + push (@read_coords, [$frag_lend, $frag_rend]); + $length_remaining -= $seg_len; + } + } + + return(@read_coords); +} + + +#### +sub _get_right_fragment_read { + my $self = shift; + my @frag = @_; + + my $read_length = $self->{read_length}; + + my $length_remaining = $read_length; + + my @read_coords; + + foreach my $segment (reverse @frag) { + my ($frag_lend, $frag_rend) = @$segment; + my $seg_len = $frag_rend - $frag_lend + 1; + + if ($seg_len >= $length_remaining) { + my $read_seg = [$frag_rend - $length_remaining + 1, $frag_rend]; + push (@read_coords, $read_seg); + last; + } + else { + push (@read_coords, [$frag_lend, $frag_rend]); + $length_remaining -= $seg_len; + } + } + + + @read_coords = reverse @read_coords; + + return(@read_coords); +} + + + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/SingleLinkageClusterer.pm b/99.scripts/trinity_utils/PerlLib/SingleLinkageClusterer.pm new file mode 100644 index 0000000..83244ed --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/SingleLinkageClusterer.pm @@ -0,0 +1,80 @@ +package main; +our $CLUSTERPATH; + + +package SingleLinkageClusterer; + +## package not to be instantiated. Just provides a namespace. + +## Input: Array containing array-refs of pairs: +## @_ = ( [1,2], [2,3], [6,7], [7,8], ...) +## Output: Array of all clusters as array-refs. +## return ([1,2,3] , [6,7,8], ...) + +use strict; + +sub build_clusters { + my @pairs = @_; + my $pairfile = "$$.pairs"; + + #must do mapping because cluster program doesn't like word chars, just ints. + my %map_id_to_feat; + my %map_feat_to_id; + my $id = 1; + + open (PAIRLIST, ">$pairfile") or die "Can't write $pairfile to /tmp"; + foreach my $pair (@pairs) { + my ($a, $b) = @$pair; + unless ($map_feat_to_id{$a}) { + $map_feat_to_id{$a} = $id; + $map_id_to_feat{$id} = $a; + $id++; + } + unless ($map_feat_to_id{$b}) { + $map_feat_to_id{$b} = $id; + $map_id_to_feat{$id} = $b; + $id++; + } + + print PAIRLIST "$map_feat_to_id{$a} $map_feat_to_id{$b}\n"; + } + close PAIRLIST; + + my $clusterfile = "$$.clusters"; + + my $cluster_prog = "slclust"; + if ($CLUSTERPATH) { + $cluster_prog = $CLUSTERPATH; + } + + system "touch $clusterfile"; + unless (-w $clusterfile) { die "Can't write $clusterfile";} + my $cmd = "$cluster_prog < $pairfile > $clusterfile"; + my $ret = system ($cmd); + if ($ret) { + die "ERROR: Couldn't run cluster properly via path: $cluster_prog.\ncmd: $cmd"; + } + + my @clusters; + open (CLUSTERS, $clusterfile); + + while (my $line = ) { + my @elements; + while ($line =~ /(\d+)\s?/g) { + push (@elements, $map_id_to_feat{$1}); + } + if (@elements) { + push (@clusters, [@elements]); + } + } + + close CLUSTERS; + + ## clean up + unlink ($pairfile, $clusterfile); + + return (@clusters); +} + + +1; diff --git a/99.scripts/trinity_utils/PerlLib/Thread_helper.pm b/99.scripts/trinity_utils/PerlLib/Thread_helper.pm new file mode 100644 index 0000000..bb43dcb --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/Thread_helper.pm @@ -0,0 +1,256 @@ +package Thread_helper; + +use strict; +use warnings; +use Carp; +use threads; + +=synopsis + + + ## here's how you might use it: + + use threads; + use Thread_helper; + + my $num_simultaneous_threads = 10; + my $thread_helper = new Thread_helper($num_simultaneous_threads); + + for (1..1000) { + + $thread_helper->wait_for_open_thread(); + + + + my $thread = threads->create(sub{ system "sleep $_";}, int(rand(20))); + $thread_helper->add_thread($thread); + } + $thread_helper->wait_for_all_threads_to_complete(); + + my @failures = $thread_helper->get_failed_threads(); + if (@failures) { + ## examine them... these are the same threads created above, use use the threads api to access info about them + ## such as error messages + } + else { + ## all good! + } + + +=cut + + +BEGIN { + + my $process_group = $$; + $SIG{'INT'} =sub { + warn "WARNING: Interrupt detected for PID group: $process_group, terminationg all child processes\n"; + system("kill -9 -$process_group"); + }; + $SIG{'KILL'} = sub { + warn "WARNING: Interrupt for PID group: $process_group, terminating all child processes.\n"; + system("kill -9 -$process_group"); + }; +} + + +my $SLEEPTIME = 1; + +our $THREAD_MONITORING = 0; # set to 1 to watch thread management + + +sub new { + my ($packagename) = shift; + my ($num_threads) = @_; + + unless ($num_threads && $num_threads =~ /^\d+$/) { + confess "Error, need number of threads as constructor param"; + } + + my $self = { + max_num_threads => $num_threads, + current_threads => [], + + error_threads => [], + + status_success => 0, + status_error => 0, + + thread_id_to_command => {}, #tid => cmd + thread_id_timing => {}, # tid => { start => val, end => val } + + + }; + + bless ($self, $packagename); + + return($self); +} + + +sub add_thread { + my $self = shift; + my ($thread, $command) = @_; + + my $thread_id = $thread->tid; + + $self->{thread_id_to_command}->{$thread_id} = $command; + $self->{thread_id_timing}->{start}->{$thread_id} = time(); + + my $num_threads = $self->get_num_threads(); + my $max_num_threads = $self->{max_num_threads}; + + if ($num_threads >= $self->{max_num_threads}) { + + # if using a wait_for_open_thread() in the client, then this condition won't be met. Best to keep it in the client rather than here. + + print STDERR "- Thread_helper: have $num_threads threads running, waiting for a thread to finish...." if $THREAD_MONITORING; + $self->wait_for_open_thread(); + print STDERR " done waiting.\n" if $THREAD_MONITORING; + } + else { + print STDERR "- Thread_helper: only $num_threads of max $max_num_threads threads running. Adding another now.\n" if $THREAD_MONITORING; + } + + push (@{$self->{current_threads}}, $thread); + + return; +} + + +#### +sub get_num_threads { + my $self = shift; + + return(scalar @{$self->{current_threads}}); +} + + +#### +sub wait_for_open_thread { + my $self = shift; + + if ($self->get_num_threads() >= $self->{max_num_threads}) { + + my $waiting_for_thread_to_complete = 1; + + my @active_threads; + + while ($waiting_for_thread_to_complete) { + + @active_threads = (); + + my @current_threads = @{$self->{current_threads}}; + foreach my $thread (@current_threads) { + if ($thread->is_running()) { + push (@active_threads, $thread); + } + else { + $waiting_for_thread_to_complete = 0; + $thread->join(); + my $status; + if (my $error = $thread->error()) { + my $thread_id = $thread->tid; + print STDERR "ERROR, thread $thread_id exited with error $error\n"; + $self->_add_error_thread($thread); + $self->{status_error}++; + $status = "ERROR"; + } + else { + $self->{status_success}++; + $status = "SUCCESS"; + } + my $thread_id = $thread->tid; + $self->{thread_id_timing}->{end}->{$thread_id} = time(); + if ($THREAD_MONITORING) { + $self->report_thread_info($thread_id, $status); + } + + } + } + if ($waiting_for_thread_to_complete) { + sleep($SLEEPTIME); + } + } + + @{$self->{current_threads}} = @active_threads; + + + } + + return; + + +} + + + +#### +sub wait_for_all_threads_to_complete { + my $self = shift; + + print STDERR "\n-now waiting for all threads to complete.\n" if $THREAD_MONITORING; + + my @current_threads = @{$self->{current_threads}}; + foreach my $thread (@current_threads) { + $thread->join(); + my $thread_id = $thread->tid; + $self->{thread_id_timing}->{end}->{$thread_id} = time(); + my $status = "SUCCESS"; + if (my $error = $thread->error()) { + print STDERR "ERROR, thread $thread_id exited with error $error\n"; + $self->_add_error_thread($thread); + $status = "ERROR"; + } + $self->{thread_id_timing}->{end}->{$thread_id} = time(); + if ($THREAD_MONITORING) { + $self->report_thread_info($thread_id, $status); + } + } + + @{$self->{current_threads}} = (); # clear them out. + + return; + +} + +#### +sub get_failed_threads { + my $self = shift; + + my @failed_threads = @{$self->{error_threads}}; + return(@failed_threads); +} + +#### +sub report_thread_info { + my $self = shift; + my ($thread_id, $status) = @_; + + my $cmd = $self->{thread_id_to_command}->{$thread_id} || "unknown"; + my $start_time = $self->{thread_id_timing}->{start}->{$thread_id}; + my $end_time = $self->{thread_id_timing}->{end}->{$thread_id}; + + print STDERR "Thread($thread_id)\t$status\tCMD: $cmd\tTime to complete: " . ($end_time-$start_time) . " seconds\n"; + + return; +} + + +############################ +## PRIVATE METHODS ######### +############################ + + +#### +sub _add_error_thread { + my $self = shift; + my ($thread) = @_; + + push (@{$self->{error_threads}}, $thread); + + return; +} + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/TiedHash.pm b/99.scripts/trinity_utils/PerlLib/TiedHash.pm new file mode 100644 index 0000000..70fd9cb --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/TiedHash.pm @@ -0,0 +1,199 @@ +#!/usr/local/bin/perl + +package TiedHash; +use strict; +use warnings; +use DB_File; +use Carp; + +=example + + my $tied_hash = new TiedHash( { create => "$pfam_db.inx" } ); + + + my $acc = ""; + + while (<$fh>) { + chomp; + my ($token, $rest) = split (/\s+/, $_, 2); + if ($token eq 'NAME') { + $acc = $rest; + } + elsif ($token =~ /^(NC|TC|DESC|ACC)$/) { + my $key = "$acc$;$token"; + $tied_hash->store_key_value($key, $rest); + print STDERR "storing: $key, $rest\n"; + } + + +=cut + + +sub new { + my $packagename = shift; + + my $prefs_href = shift; + + if ($prefs_href && ! ref $prefs_href) { + confess "Error, need hash reference with opts in constructor.\n"; + } + + + my $self = { + index_filename => undef, + tied_index => {}, + tie_invoked => 0, + }; + + bless ($self, $packagename); + + + if (ref $prefs_href eq "HASH") { + if (my $index_file = $prefs_href->{"create"}) { + $self->create_index_file($index_file); + } + elsif ($index_file = $prefs_href->{"use"}) { + $self->use_index_file($index_file); + } + } + + + return ($self); +} + +#### +sub tie_invoked { + my $self = shift; + return ($self->{tie_invoked}); +} + + +#### +sub DESTROY { + my $self = shift; + if ($self->{index_filename}) { + # hash must have been tied + # so, untie it + untie (%{$self->{tied_index}}); + } +} + + +#### +sub create_index_file { + my $self = shift; + return ($self->make_index_file(@_)); +} + + + +#### +sub make_index_file { + my $self = shift; + my $filename = shift; + + unless ($filename) { + confess "need filename as parameter"; + } + + if (-e $filename) { + unlink $filename or confess "cannot remove existing index filename $filename"; + } + + $self->{index_filename} = $filename; + + tie (%{$self->{tied_index}}, 'DB_File', $filename, O_CREAT|O_RDWR, 0666, $DB_BTREE); + + $self->{tie_invoked} = 1; + + return; +} + + +#### +sub use_index_file { + my $self = shift; + my $filename = shift; + + unless ($filename) { + confess "need filename as parameter"; + } + + unless (-s $filename) { + confess "Error, cannot locate file: $filename\n"; + } + + $self->{index_filename} = $filename; + + tie (%{$self->{tied_index}}, 'DB_File', $filename, O_RDONLY, 0, $DB_BTREE); + + $self->{tie_invoked} = 1; + + #my @keys = $self->get_keys(); + #unless (@keys) { + # confess "Error, tried using $filename db, but couldn't perform retrievals.\n"; + #} + + return; + +} + + +#### +sub store_key_value { + my ($self, $identifier, $value) = @_; + + #my $num_keys = scalar ($self->get_keys()); + + unless ($self->tie_invoked()) { + confess "Error, cannot store key/value pair since tied hash not created.\n"; + } + + + my $found = 0; + while (! $found) { + $self->{tied_index}->{$identifier} = $value; + + my $val = $self->get_value($identifier); + if (defined $val) { + $found = 1; + } + else { + warn "Berkeley DB had trouble storing ($identifier); trying again.\n"; + } + } + + return; + +} + + +#### +sub get_value { + my $self = shift; + my $identifier = shift; + + + unless ($self->tie_invoked()) { + confess "Error, cannot retrieve value from untied hash\n"; + } + + my $value = $self->{tied_index}->{$identifier}; + + return ($value); +} + + +## +sub get_keys { + my $self = shift; + + unless ($self->tie_invoked()) { + confess "Error, cannot retrieve values from untied hash\n"; + } + + return (keys %{$self->{tied_index}}); +} + + +1; #EOM diff --git a/99.scripts/trinity_utils/PerlLib/VCF_parser.pm b/99.scripts/trinity_utils/PerlLib/VCF_parser.pm new file mode 100644 index 0000000..f02af8c --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/VCF_parser.pm @@ -0,0 +1,173 @@ +package VCF_parser; + +use strict; +use warnings; + +use Carp; + +sub new { + my ($packagename) = shift; + my ($filename) = @_; + + unless ($filename) { + confess "Error, need filename as parameter"; + } + + my $self = { filename => $filename, + fh => undef, + }; + + bless ($self, $packagename); + + $self->_init(); + + return($self); +} + +#### +sub _init { + my ($self) = @_; + + my $filename = $self->{filename}; + open (my $fh, $filename) or confess "Error, cannot open file $filename"; + + $self->{fh} = $fh; + + return; +} + + +#### +sub get_next { + my $self = shift; + + my $fh = $self->{fh}; + my $line = <$fh>; + while ($line && $line =~ /^\#/) { + # skip the header lines + $line = <$fh>; + } + + if ($line) { + return(VCF_record->new($line)); + } + else { + return(undef); + } +} + + + + +#################################### +#################################### + +package VCF_record; + +use strict; +use warnings; + +use Carp; + +sub new { + my ($packagename) = shift; + my ($vcf_line) = @_; + + unless ($vcf_line =~ /\w/) { + confess "Error, require vcf line of text as parameter"; + } + + my $struct = &_parse_vcf_line($vcf_line); + + bless($struct, $packagename); + + return($struct); +} + +#### +sub _parse_vcf_line { + my ($vcf_line) = @_; + + chomp $vcf_line; + + my @x = split(/\t/, $vcf_line); + + my $acc = $x[0]; + my $pos = $x[1]; + my $ref_base = $x[3]; + my $allele_base = $x[4]; + + my $tag_info = $x[7]; + + my %tags; + + foreach my $keyval_pair (split(/;/, $tag_info)) { + + if ($keyval_pair =~ /=/) { + my ($key, $val) = split(/=/, $keyval_pair); + $tags{$key} = $val; + } + } + + my $struct = { line => $vcf_line, + + acc => $acc, + pos => $pos, + + ref_base => $ref_base, + allele_base => $allele_base, + tag_info => $tag_info, + tags_href => \%tags, + }; + + + return($struct); +} + + +#### +sub get_accession { + my ($self) = @_; + return($self->{acc}); +} + + +#### +sub get_position { + my ($self) = @_; + return($self->{pos}); +} + +#### +sub get_ref_base { + my ($self) = @_; + return($self->{ref_base}); +} + +#### +sub get_allelic_base { + my ($self) = @_; + return($self->{allele_base}); +} + +#### +sub get_tag_val { + my ($self) = shift; + my ($tagname) = @_; + + return($self->{tags_href}->{$tagname}); +} + +#### +sub has_tag { + my ($self) = shift; + my ($tagname) = @_; + + return(exists $self->{tags_href}->{$tagname}); +} + + + + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/WigParser.pm b/99.scripts/trinity_utils/PerlLib/WigParser.pm new file mode 100644 index 0000000..8fec870 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/WigParser.pm @@ -0,0 +1,202 @@ +package WigParser; + +use strict; +use warnings; +use Carp; + +#### +sub new { + my $packagename = shift; + my ($wig_file) = @_; + + unless ($wig_file) { + confess "Error, need wig_filename as parameter to constructor"; + } + + my $self = { wig_file => $wig_file, + wig_file_fh => undef, + contig_to_seek_pos_href => {}, + }; + + bless ($self, $packagename); + + $self->_init_wig_info(); + + + return($self); +} + +#### +sub _init_wig_info { + my $self = shift; + + my $wig_file = $self->{wig_file}; + + + open (my $fh, $wig_file) or die "Error, cannot open file $wig_file"; + $self->{wig_file_fh} = $fh; + + while (<$fh>) { + if (/chrom=(\S+)/) { + my $scaff = $1; + my $filepos = tell($fh); + $self->{contig_to_seek_pos_href}->{$scaff} = $filepos; + } + } + + + return; +} + +#### +sub get_contig_list { + my $self = shift; + + my @contigs = keys %{$self->{contig_to_seek_pos_href}}; + + return(@contigs); +} + +#### +sub get_wig_array { + my $self = shift; + my ($contig, $extended_flag) = @_; + + + my $seekpos = $self->{contig_to_seek_pos_href}->{$contig}; + + unless (defined $seekpos) { + confess "Error, cannot find seek position entry for contig: $contig"; + } + + my $fh = $self->{wig_file_fh}; + + seek($fh, $seekpos, 0); + + #print STDERR "-retrieving wig array for $contig at pos: $seekpos\n"; + + my @wig_array; + + while (<$fh>) { + + if (/chrom=/) { last; } + + chomp; + my ($pos, $val, @rest) = split(/\s+/); + if ($pos > 1e12) { + confess "Error, position $pos is out of range of max contig position value set at 1e12"; + } + if ($extended_flag) { + $wig_array[$pos] = [$val, @rest]; + } + else { + $wig_array[$pos] = $val; + } + + } + + # fill in missing entries + foreach my $val (@wig_array) { + if (! defined ($val)) { + $val = 0; + } + } + + + return(@wig_array); +} + + + +##################################################################### +## Static method that uses LOTS of memory (original implementation) + +sub parse_wig { + my ($wig_file) = @_; + + my %scaff_to_coverage; + + print STDERR "-retrieving max positions per scaffold\n"; + my %max_vals_for_scaffs = &_parse_max_scaff_lengths($wig_file); + + print STDERR "-preallocating memory for coverage\n"; + ## preallocate arrays for coverage + + my $sum_pos = 0; + foreach my $scaff (keys %max_vals_for_scaffs) { + my $max_val = $max_vals_for_scaffs{$scaff}; + + my @cov_vals = (); + $#cov_vals = $max_val; + + for my $index (0..$max_val) { + $cov_vals[$index] = int(0); + } + $scaff_to_coverage{$scaff} = \@cov_vals; + + $sum_pos += $max_val; + + } + + print STDERR "- $sum_pos bases represented\n"; + + print STDERR "-populating coverage data into memory.\n"; + + my $scaff = undef; + + open (my $fh, $wig_file) or die "Error, cannot open file $wig_file"; + while (<$fh>) { + if (/^track/) { next; }; + if (/chrom=(\S+)/) { + $scaff = $1; + next; + } + chomp; + my ($pos, $val) = split(/\s+/); + $scaff_to_coverage{$scaff}->[$pos] = $val; + + } + + close $fh; + + return(%scaff_to_coverage); +} + + + +### Private + +sub _parse_max_scaff_lengths { + my ($wig_file) = @_; + + my %scaff_to_max_vals; + + my $scaff = ""; + + + my $counter = 0; + open (my $fh, $wig_file) or die "Error, cannot open file $wig_file"; + while (<$fh>) { + if (/^track/) { next; }; + if (/chrom=(\S+)/) { + $scaff = $1; + next; + } + chomp; + my ($pos, $val) = split(/\s+/); + $scaff_to_max_vals{$scaff} = $pos; + + #print "scaff: $scaff\t$pos=> $val\n"; + + $counter++; + + #if ($counter > 100000) { last; } + } + + close $fh; + + return(%scaff_to_max_vals); +} + +1; #EOM + diff --git a/99.scripts/trinity_utils/PerlLib/overlapping_nucs.ph b/99.scripts/trinity_utils/PerlLib/overlapping_nucs.ph new file mode 100644 index 0000000..eea9696 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/overlapping_nucs.ph @@ -0,0 +1,61 @@ +#!/usr/local/bin/perl + +#### +sub nucs_in_common { + my ($e5, $e3, $g5, $g3) = @_; + + + ($e5, $e3) = sort {$a<=>$b} ($e5, $e3); + ($g5, $g3) = sort {$a<=>$b} ($g5, $g3); + + unless (&coordsets_overlap([$e5,$e3], [$g5, $g3])) { + return(0); + } + + my $length = abs ($e3 - $e5) + 1; + my $diff1 = ($e3 - $g3); + $diff1 = ($diff1 > 0) ? $diff1 : 0; + my $diff2 = ($g5 - $e5); + $diff2 = ($diff2 > 0) ? $diff2 : 0; + my $overlap_length = $length - $diff1 - $diff2; + return ($overlap_length); +} + +#### +sub coordsets_overlap { + my ($coordset_A_aref, $coordset_B_aref) = @_; + + my ($lend_A, $rend_A) = sort {$a<=>$b} @$coordset_A_aref; + + my ($lend_B, $rend_B) = sort {$a<=>$b} @$coordset_B_aref; + + if ($lend_A <= $rend_B && $rend_A >= $lend_B) { + ## yes, overlap + return (1); + } + else { + return (0); + } +} + + +#### +sub coordset_A_encapsulates_B { + my ($coordset_A_aref, $coordset_B_aref) = @_; + + my ($lend_A, $rend_A) = sort {$a<=>$b} @$coordset_A_aref; + + my ($lend_B, $rend_B) = sort {$a<=>$b} @$coordset_B_aref; + + if ($lend_A <= $lend_B && $rend_B <= $rend_A) { + return(1); # true + } + else { + return(0); # false; + } +} + + + +1; + diff --git a/99.scripts/trinity_utils/PerlLib/test_Fasta_retriever.pl b/99.scripts/trinity_utils/PerlLib/test_Fasta_retriever.pl new file mode 100644 index 0000000..6da2050 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/test_Fasta_retriever.pl @@ -0,0 +1,39 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Fasta_retriever; +use List::Util qw(shuffle); + +my $usage = "usage: $0 file.fasta\n\n"; + +my $file = $ARGV[0] or die $usage; + +main: { + + my @accs; + open (my $fh, $file) or die $!; + while (<$fh>) { + if (/^>(\S+)/) { + my $acc = $1; + push (@accs, $acc); + } + } + close $fh; + + + @accs = shuffle @accs; + + my $fasta_retriever = new Fasta_retriever($file); + + foreach my $acc (@accs) { + + my $seq = $fasta_retriever->get_seq($acc); + + print "$acc\t$seq\n"; + } + + exit(0); +} + diff --git a/99.scripts/trinity_utils/PerlLib/test_htc_gridrunner_LSF.pl b/99.scripts/trinity_utils/PerlLib/test_htc_gridrunner_LSF.pl new file mode 100644 index 0000000..0d770a4 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/test_htc_gridrunner_LSF.pl @@ -0,0 +1,27 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::RealBin"); + +use HTC::GridRunner; + +my $config_file = "$FindBin::RealBin/../htc_conf/BroadInst_LSF.test.conf"; + +main: { + + my @cmds; + for my $num (1..10) { + my $cmd = "echo hello $num"; + push (@cmds, $cmd); + } + push (@cmds, "this_command_should_fail"); + + my $grid_runner = new HTC::GridRunner($config_file, "cache_completed_LSF_cmds"); + + my $ret = $grid_runner->run_on_grid(@cmds); + + exit($ret); +} + diff --git a/99.scripts/trinity_utils/PerlLib/test_htc_gridrunner_SGE.pl b/99.scripts/trinity_utils/PerlLib/test_htc_gridrunner_SGE.pl new file mode 100644 index 0000000..a0f78f3 --- /dev/null +++ b/99.scripts/trinity_utils/PerlLib/test_htc_gridrunner_SGE.pl @@ -0,0 +1,29 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::RealBin"); + +use HTC::GridRunner; + +my $config_file = "$FindBin::RealBin/../htc_conf/BroadInst_SGE.test.conf"; + +main: { + + my $cache_file = "cache_completed_SGE_cmds"; + + my @cmds; + for my $num (1..10) { + my $cmd = "echo hello $num"; + push (@cmds, $cmd); + } + push (@cmds, "this_command_should_fail"); + + my $grid_runner = new HTC::GridRunner($config_file, $cache_file); + + my $ret = $grid_runner->run_on_grid(@cmds); + + exit($ret); +} + diff --git a/99.scripts/trinity_utils/util/PBS/N50stats.pl b/99.scripts/trinity_utils/util/PBS/N50stats.pl new file mode 100644 index 0000000..3b6e8d3 --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/N50stats.pl @@ -0,0 +1,291 @@ +#!/usr/bin/env perl + +=pod + +=head1 NAME + +=head1 USAGE + + -in input file in FASTA/Q or posmap length file (e.g. posmap.scflen) + -genome genome size in bp for estimating N lengths and indexes + -single FASTA/Q has sequence in a single line (faster) + -overwrite => Force overwrite + -reads => Force processing as read data (no N50 statistics) + -noreads => Force as not being read data. Good for cDNA assemblies with short contigs + +=head1 AUTHORS + + Alexie Papanicolaou 1 + + Ecosystem Sciences, CSIRO, Black Mountain Labs, Clunies Ross Str, Canberra, Australia + alexie@butterflybase.org + +=head1 DISCLAIMER & LICENSE + +This software is released under the GNU General Public License version 3 (GPLv3). +It is provided "as is" without warranty of any kind. +You can find the terms and conditions at http://www.opensource.org/licenses/gpl-3.0.html. +Please note that incorporating the whole software or parts of its code in proprietary software +is prohibited under the current license. + +=head1 BUGS & LIMITATIONS + +None known so far. + +=cut + +use strict; +use warnings; +use Getopt::Long; +use Pod::Usage; +use Statistics::Descriptive; +use Bio::SeqIO; +$|=1; + +my (@infiles,$user_genome_size,$is_fasta,$is_fastq,$is_single,$overwrite,$is_reads,$isnot_reads); +GetOptions( + 'in=s{,}' => \@infiles, + 'single' =>\$is_single, + 'genome:s' => \$user_genome_size, + 'overwrite' => \$overwrite, + 'reads' =>\$is_reads, + 'noreads' =>\$isnot_reads, +); +if (!@infiles){ + @infiles = @ARGV; +} +pod2usage "No input files!\n" if !@infiles; +die "Cannot ask for both reads and noreads options at the same time!\n" if $is_reads && $isnot_reads; + +if ($is_reads && !$isnot_reads){ + print "Processing all data as reads\n"; +} +if ($user_genome_size && $user_genome_size=~/\D/){ + if ($user_genome_size=~/^(\d+)kb$/i){ + $user_genome_size=int($1.'000'); + } + elsif ($user_genome_size=~/^(\d+)mb$/i){ + $user_genome_size=int($1.'000000'); + } + elsif ($user_genome_size=~/^(\d+)gb$/i){ + $user_genome_size=int($1.'000000000'); + } + print "Genome set to ".&thousands($user_genome_size)." b.p.\n"; +} + +foreach my $infile (@infiles){ + unless ($infile && -s $infile){warn("I need a posmap length file, e.g. .posmap.scflen for scaffolds\n");pod2usage;} + my $outfile=$infile.'.n50'; + $outfile.='g' if ($user_genome_size); + warn ("Outfile $outfile already exists\n") if -s $outfile && !$overwrite; + next if -s $outfile && !$overwrite; + my $total=int(0); + my $gaps = int(0); + my $seq_ref; + my @head=`head $infile`; + foreach (@head){ + if ($_=~/^>\S/){ + $is_fasta=1; + print "FASTA file found!\n"; + last; + }elsif($_=~/^@\S/){ + $is_fastq=1; + print "FASTQ file found!\n"; + last; + } + } + + print "Parsing file $infile...\n"; + if ($is_fasta){ + ($total,$gaps,$seq_ref) = &process_fasta($infile); + } + elsif($is_fastq){ + ($total,$gaps,$seq_ref) = &process_fastq($infile); + } + else { + ($total,$gaps,$seq_ref) = &process_csv($infile); + } + print "Preparing stats...\n"; + my ($mean,$n50,$n10,$n25,$n50_length,$n10_length,$n25_length,$scaffolds,$scaffolds_size,$smallest,$largest,$sequence_number,$sum,$genome_size) = &process_stats($seq_ref,$total); + + if ($mean){ + open (OUT,">".$outfile); + my $stat = Statistics::Descriptive::Full->new(); + $stat->add_data($seq_ref); + #my $skew='';sprintf("%.2f",$stat->skewness()); + my $mean = sprintf("%.2f",$mean); + my $median = $stat->median(); + my $var = sprintf("%.2f",$stat->variance()); + my $sd = sprintf("%.2f",$stat->standard_deviation()); + if (!$scaffolds || $scaffolds == 0){ + $scaffolds=$sequence_number; + $scaffolds_size=$smallest; + } + print OUT "File: $infile\n"; + print OUT "TOTAL: ".&thousands($total)." bp in ".&thousands($sequence_number)." sequences\n"; + print OUT "\tof which ".&thousands($gaps)." are Ns/gaps.\n"; + print OUT "Mean: ".&thousands($mean)."\nStdev: ".&thousands($sd)."\n"; + print OUT "Median: ".&thousands($median)."\n"; + print OUT "Smallest: ".&thousands($smallest)."\nLargest: ".&thousands($largest)."\n"; + if (($mean >=1000 && !$is_reads) || $isnot_reads){ + print OUT "N10 length: ".&thousands($n10_length)."\nN10 Number: ".&thousands($n10)."\n"; + print OUT "N25 length: ".&thousands($n25_length)."\nN25 Number: ".&thousands($n25)."\n"; + print OUT "N50 length: ".&thousands($n50_length)."\nN50 Number: ".&thousands($n50)."\n"; + print OUT "Assuming a genome size of " + .&thousands($user_genome_size) + ." then the top ".&thousands($scaffolds) + ." account for it (min " + .&thousands($scaffolds_size) + ." bp)\n" if $user_genome_size; + }else{ + print OUT "Reads found! Read coverage estimated to ".sprintf("%.2f",$total/$user_genome_size)."x using user provided genome size of ".&thousands($user_genome_size)."\n" if $user_genome_size; + } + close (OUT); + print "Done, see $outfile\n"; + system("cat $outfile"); + }else { + open (OUT,">".$outfile); + print OUT "File: $infile\n"; + print OUT "TOTAL: $total bp in $sequence_number sequences\n"; + close (OUT); + warn "Non fatal warning: Something went wrong in estimating the statistics. Maybe the provided genome length is much larger than sequence length or maybe less than 3 sequences provided?\n"; + } +} +######################################################################## +sub process_fasta(){ + print "Processing as FASTA\n"; + my $infile=shift; + my @array ; + my $total=int(0); + my $gaps=int(0); + my $counter = int(0); + + if ($is_single){ + open (IN,$infile)||die; + while (my $seq_id=) { + my $seq=; + my $length=length($seq)-1; # newline + $counter+=length($seq_id)+$length+1; + next unless $length; + $gaps+=($seq=~tr/[N\-]//); + push(@array,$length); + $total+=$length; + } + close IN; + }else{ + my $filein = new Bio::SeqIO(-file=>$infile , -format=>'fasta'); + while (my $seq_obj=$filein->next_seq()) { + $counter+=length($seq_obj->seq().$seq_obj->description().' '.$seq_obj->id()) if $seq_obj->seq(); + my $length=$seq_obj->length(); + my $seq=$seq_obj->seq(); + $gaps+=($seq=~tr/[N\-]//); + next unless $length; + push(@array,$length); + $total+=$length; + } + } + print "\n"; + die "No data found or wrong format\n" unless $total; + return ($total,$gaps,\@array); +} +sub process_fastq(){ + print "Processing as FASTQ\n"; + my $infile=shift; + my @array ; + my $total=int(0); + my $gaps=int(0); + my $counter = int(0); + open (IN,$infile); + while (my $seq_id=) { + my $seq=; + my $scrap=.; + my $length=length($seq)-1; #newline + $counter+=length($seq_id)+$length+1; + next unless $length; + $gaps+=($seq=~tr/[N\-]//); + push(@array,$length); + $total+=$length; + } + close IN; + print "\n"; + die "No data found or wrong format\n" unless $total; + return ($total,$gaps,\@array); +} +sub process_csv(){ + print "Processing as CSV\n"; + my $infile=shift; + my @array ; + my $total=int(0); + my $gaps='N/A'; + my $counter = int(0); + open (IN,$infile)||die ("Cannot open $infile\n"); + while (my $ln=){ + $counter+=length($ln); + $ln=~/(\d+)$/; + next unless $1; + my $length= $1; + push(@array,$1); + $total+=$length; + } + close IN; + die "No data found or wrong format\n" unless $total; + return ($total,$gaps,\@array); +} + +sub process_stats(){ + my $sequences_ref = shift; + my $total = shift; + my $genome_size = int(0); + my $mean = $total / scalar(@$sequences_ref); + my ($n50,$n10,$n25,$n50_length,$n10_length,$n25_length,$scaffolds,$scaffolds_size,$smallest,$largest,$sequence_number,$sum); + print "Sorting..."; + my @sequences=sort{$b<=>$a} @$sequences_ref; + $smallest=$sequences[-1]; + $largest=$sequences[0]; + print " done!\n"; + $|=0; + if (($mean < 1000 && !$isnot_reads) || $is_reads ){ + print "Reads detected. Ignoring N* calculations.\n"; + $genome_size=$total; + $sequence_number = scalar(@$sequences_ref); + return ($mean,$n50,$n10,$n25,$n50_length,$n10_length,$n25_length,$scaffolds,$scaffolds_size,$smallest,$largest,$sequence_number,$sum) if $mean <1000; + } + elsif (!$user_genome_size){ + print "Setting genome size for N* calculations to total consensus $total\n"; + $genome_size=$total; + }elsif($user_genome_size){ + print "Setting genome size for N* calculations to user defined $user_genome_size\n"; + $genome_size = $user_genome_size; + } + + foreach my $sequence_length ( @sequences){ + $sum+=$sequence_length; + $sequence_number++; + if($sum >= $genome_size*0.1 && !$n10){ + $n10=$sequence_number; + $n10_length=$sequence_length; + } + elsif($sum >= $genome_size*0.25 && !$n25){ + $n25=$sequence_number; + $n25_length=$sequence_length; + } + elsif($sum >= $genome_size*0.5 && !$n50){ + $n50 = $sequence_number; + $n50_length=$sequence_length; + }elsif ($sum >= $genome_size && !$scaffolds){ + $scaffolds = $sequence_number; + $scaffolds_size = $sequence_length; + } + + } + print "Processed $sequence_number sequences\n"; + return ($mean,$n50,$n10,$n25,$n50_length,$n10_length,$n25_length,$scaffolds,$scaffolds_size,$smallest,$largest,$sequence_number,$sum,$genome_size); +} + +sub thousands($){ + my $val = shift; + return int(0) if !$val; + $val = sprintf("%.0f", $val); + 1 while $val =~ s/(.*\d)(\d\d\d)/$1,$2/; + return $val; +} diff --git a/99.scripts/trinity_utils/util/PBS/README b/99.scripts/trinity_utils/util/PBS/README new file mode 100644 index 0000000..0087b57 --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/README @@ -0,0 +1,58 @@ +################################################################################################################ +######################## README file ######################## +######################## Trinity PBS job submission with multi part dependencies ######################## +######################## Author: Josh Bowden, Alexie Papanicolaou, CSIRO ######################## +######################## Email: alexie@butterflybase.org ######################## +######################## Version 1.0 ######################## +################################################################################################################ + +DESCRIPTION: BASH shell scripts for submission of Trinity jobs to clusters that use PBS Torque or PBS Pro. +The set of scripts stages the parts of the Trinity workflow into 6 stages: + 1/ Inchworm + 2/ Chrysalis::GraphFromFasta and Chrysalis::ReadsToTranscripts if walltime permits + 3/ Chrysalis::ReadsToTranscripts + 4/ Chrysalis::QuantifyGraph + 5/ Butterfly + 6/ Gather together resulting transcripts into "Trinity.fasta" file + +Each stage is submitted as a PBS job, with dependencies i.e. the following stage will only execute after successfull completion of the stage before. +Due to their parallel nature, stages 4 and 5 are submitted as 'array jobs', with each job made up of a user defined number of subtasks. + +ADMINISTRATION setup instructions: + trinity_pbs script install instructions: + 1. Copy all trinity_pbs.* files into a directory (we will call it "TRINITY_PBS_DIR") in a user accessible, read only, area. + 2. Add TRINITY_PBS_DIR to the PATH i.e. export or set PATH=TRINITY_PBS_DIR:$PATH + 3. Make sure trinity_pbs.sh is executable (chmod 755 trinity_pbs.sh) + In the file trinity_pbs.sh, do the following: + 4. Change the "TRINITYPATH" variable to point to the trinity.pl installation directory. + 5. Set MEMDIRIN to the name of a node-local filesystem so a network drive is not needed for final parallel stages + 6. Set MODTRINITY so that the trinity executables will be available - load approprite modules and set PATH. + 7. Set PBSTYPE to --pbspro or --pbs, dependent on the system present. + That should be all that is needed from an admin perspective. These scripts have been tested on PBS Torque 3.0.6 and PBS Pro 11.0.2. + There may be PBS system specific changes due to PBS version incompatabilities. + 8. Maybe make any system-specific changes to the user-specific text README below or TRINITY.CONFIG.template before distributing + +USER setup instructions: + 1. Users should make a copy of TRINITY.CONFIG.template (possibly a copy for each job they want to run) and then modify variables in it + to suit the system the job is being run on (number of CPUs, amount of memory) and the expected runtime of the Trinity process + (which is a function of the datset size). Further instructions are provided within the TRINITY.CONFIG.template file. + 2. While it is running, you can use qsub -u $USER to see your jobs + 3. We provide a script, pbs_check.pl, that allows you to see the progress of your jobs; just give the job id + 4. If any step fails, then some jobs that depended on a /successful/ completion of that step will be stuck in a HOLD (H) status + 5. The script trinity_kill.pl can be used to kill all running, queued or held jobs by passing it the output data directory (OUTPUTDIR). + 6. When you re-submit a job with trinity_pbs.sh , trinity_kill.pl will automatically run and stop and jobs in the directory specified by + your config file + (we take no responsibility if you delete the chrysalis directory and Trinity.fasta does not have all the data - e.g. because the PBS crashed) + + +USER recommendations + * Walltimes depend on how much data you have and also on your local HPC environment well (e.g. speed of I/O - hard disks and CPU configuration). + In the beginning you may want to err towards higher walltimes, ask colleagues using the same machines. The default values worked well for us. + * For quantify graph/Butterfly (steps 4/5), we start with a particular walltime (say 2h). Often not all jobs complete. Because it is a batch of many commands, + the best approach is to simply re-launch trinity_pbs and it will continue to process the rest of the commands. Check the logfile output from the PBS + to see if the job fails because one particular quantifygraph or Butterfly job takes longer than the given walltime (in that case: increase the walltime, run the command + manually or decrease the number of reads used - -max_reads for quantifygraph). It is possible that it is a very long gene, a bacterial contaminant or + some other oddity that is delaying your assembly. + * Do not overload the I/O of the system (steps 4/5) by starting a lot of jobs (i.e. multiple trinity assemblies). In that case, increase the + number of commands run within each step 4/5 batch (NUMPERARRAYITEM variable in the CONFIG) + * NB: Always count how many sequences you get at the end (Trinity.fasta) to make sure that the script has completed as you expected. diff --git a/99.scripts/trinity_utils/util/PBS/TRINITY.CONFIG.template b/99.scripts/trinity_utils/util/PBS/TRINITY.CONFIG.template new file mode 100644 index 0000000..d2dde06 --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/TRINITY.CONFIG.template @@ -0,0 +1,122 @@ +################################################################################################################################## +########################## ######################################## +########################## Trinity PBS job submission with multi part dependencies ######################################## +########################## User modifiyable input file ######################################## +########################## ######################################## +################################################################################################################################## +### Author: Josh Bowden, CSIRO IM&T, Alexie Papanicolaou CSIRO CES +### Email: alexie@butterflybase.org +### Version 1.0 +### +### Configuration file for script to split the Trinity workflow into multiple stages so as to efficiently request +### and use appropriate resources (walltime and number of cores) on a computer cluster / supercomputer. +### +### User must set all the variables in this file to appropriate values +### and then run the trinity_pbs.sh script with this file as input as shown below: +### +### Command line usage: +### To start (or re-start) an analysis: +### >trinity_pbs.sh TRINITY.CONFIG.template +### To stop previously started PBS jobs on the queue: +### >trinity_pbs.sh --rm OUTPUTDIR +### Where: +### TRINITY.CONFIG.template = user specific job details (i.e. the current file) +### OUTPUTDIR = is path to output data directory (set below) +### +### If any stage fails, the jobs may be resubmitted and only the scripts that have not completed +### will be resubmitted to the batch system. Either the scripts can be re-run (by using trinity_pbs.sh) +### with original (or new) inputs from the current file (changing the variables below) +### or the original scripts created by trinity_pbs.sh can be re-run by finding them in the output directory. +### +### Each stage and each array job will have a PBS output file sent to the output directory when the job finishes. +### This means there will be many output files from the PBS system when the array job runs (from part 4 and 5 mostly). +### If any part fails, errors will be specified in these output files. +### +### N.B. The trinity_pbs.sh file must have a number of system specific variables set by a system administrator +### +################################################################################################################################## + +# USER must edit these: +###### Set an email to which job progress and status will be sent to. +UEMAIL= +###### Set a valid account (if available), otherwise leave blank. +ACCOUNT="#PBS -A sf-CSIRO" +###### Select a value for JOBPRFIX that is not longer than 7 characters. NO spaces or other non-alphanumeric characters +JOBPREFIX= + +###### Set output data directory (OUTPUTDIR) +###### OUTPUTDIR is where PBS scripts will be written and also Trinity results will be stored +###### This area requires a large amount of space (possibly 100's of GB) and a high file count +###### ($WORKDIR is a standard area on some systems, however users should check that it is valid on their machine) +OUTPUTDIR="$WORKDIR"/trinityrnaseq/"$JOBPREFIX" + +###### Set input data directory. This has to be explicitly set as it is used in other internal scripts. +###### Make sure you include the final forward slash. Defaults to current directory +DATADIRECTORY=$PWD/ +###### Set input filenames, you can use wildcards if you embed the filename is 'single quotes' +FILENAMELEFT='*_left.fasta' # change this +FILENAMERIGHT='*_right.fasta' #change this +FILENAMESINGLE=single.fasta # change this or set it to empty +SEQTYPE=fa # change this to fa (FASTA) fq (FASTQ) or cfa/cfq (FASTA or FASTQ colour space SOLiD ABI) + + +# User may opt to change these: +# we set --max_reads_per_graph to 1million because very high I/O is needed otherwise. It is unlikely that a transcript needs more than 1 million reads to be assembled.... +FILENAMEINPUT=" --seqType "$SEQTYPE" --left "$DATADIRECTORY""$FILENAMELEFT" --right "$DATADIRECTORY""$FILENAMERIGHT" --max_reads_per_graph 1000000 " +### STANDARD_JOB_DETAILS sets analysis specific input to Trinity.pl +### N.B. do not use --CPU or JM flag as this is automatically appended +STANDARD_JOB_DETAILS="Trinity.pl "$FILENAMEINPUT" --output "$OUTPUTDIR"" + + +# This is where you specify resource limits. We provide some defaults values +# Ultimately settings depends on your data size and complexity +# Steps that go beyond their walltime will not complete. Edit these values and resubmit +### Stage P1: Time and resources required for Inchworm stage +### Only use at maximum, half the available CPUs on a node +# - Inchworm will not efficiently use any more than 4 CPUs and you will have to take longer for resources to be assigned +WALLTIME_P1="2:00:00" +MEM_P1="20gb" # will use it for --JM +NCPU_P1="4" +PBSNODETYPE_P1="any" # ask you system administrator what Nodetypes exists + +### Stage P2: Time and resources required for Chrysalis stage +### Starts with Bowtie alignment and post-processing of alignment file +### All CPUs presenct can be used for the Chrysalis parts. +#They may take a while to be provisioned, so the less request, possibly the faster the jobs turnaround. +# For one step (the parallel sort) it needs as much memory as specified in P1. Less memory, means more I/O for sorting +# increase for more lanes of data: 3 lanes-> 60gb of RAM and 24h of time will do it. +# The bowtie step will take considerable amount of time with more data +WALLTIME_P2="12:00:00" +MEM_P2="20gb" # will use it for the parallel sort of the SAM after alignment +NCPU_P2="6" +PBSNODETYPE_P2="any" + +### Stage P3: This is a backup stage for Chrysalis - only runs if time ran out in P2 above. +### This will need about 1 day per lane +WALLTIME_P3="18:00:00" +MEM_P3="8gb" +NCPU_P3="6" +PBSNODETYPE_P3="medium" + +### Stage P4: QuantifyGraph graph runs in many parallel parts +### Tasks that fail will reamain in the OUTPUTDIR/chrysalis/quantifyGraph_commands.XYZ files (XYZ is a number) +### The remaining tasks can be run by running the job submission command "trinity_pbs.sh " again. +NUMPERARRAYITEM_P4=5000 +WALLTIME_P4="00:30:00" +MEM_P4="4gb" +NCPU_P4="1" +PBSNODETYPE_P4="medium" + +### Stage P5: Butterfly options. Some butterfly jobs can take exceedingly long. Users may need to restart trinity_pbs multiple times to complete. +### Tasks that fail will reamain in the OUTPUTDIR/chrysalis/butterfly_commands.adj.XYZ files (XYZ is a number) +### Often there will be a few tasks that take a lot longer than others so multiple submissions to the cluster may be required. +### The remaining tasks can be run by running the job submission command "trinity_pbs.sh " again. +### Running the long running tasks seperately may also be a good option. +NUMPERARRAYITEM_P5=5000 +WALLTIME_P5="01:00:00" +MEM_P5="10gb" +NCPU_P5="1" +PBSNODETYPE_P5="medium" + + + diff --git a/99.scripts/trinity_utils/util/PBS/pbs_check.pl b/99.scripts/trinity_utils/util/PBS/pbs_check.pl new file mode 100644 index 0000000..a5e14dd --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/pbs_check.pl @@ -0,0 +1,100 @@ +#!/usr/bin/env perl + +=pod + +=head1 NAME + + Monitor a PBS job progress + +=head1 USAGE + + [options] + + Options: + one of + -top use top on execution host + -dump dump output or error (default) + -follow follow output/error + -tail tail of output/error (the last 10 lines) + -head head of output/error (the first 10 lines) + + and also + -e|error Show stderr instead of stdout + -s|spool Location of spool directory (defaults to /var/spool/PBS/spool) + +=head1 LICENSE + + Released under the MIT License - Alexie Papanicolaou 2012, CSIRO Ecosystem Sciences, alexie@butterflybase.org + +=cut + + +use strict; +use warnings; +use Pod::Usage; +use Getopt::Long; +my ($head,$follow,$tail,$dump,$show_error,$spool_dir,$top); +GetOptions( + 'top' => \$top, + 'head' => \$head, + 'follow' => \$follow, + 'tail' => \$tail, + 'cat|dump' => \$dump, + 's|spool_dir:s' => \$spool_dir, + 'error' => \$show_error, +); + +my $method; +if ($top){ + $method = 'top'; +}elsif ($head){ + $method = 'head'; +}elsif ($follow){ + $method = 'follow' +}elsif ($tail){ + $method = 'tail'; +}else{ + $method = 'cat'; +} + +my $jobid = shift; + +pod2usage unless $jobid; + +$spool_dir = $spool_dir ? $spool_dir : '/var/spool/PBS/spool'; # exists on exec host but necessarily on submit host +my $user = $ENV{'USER'}; +my $exec = 'ssh '; +pod2usage "No job ID provided\n" unless $jobid; +$jobid=~/^(\w+\[?\d*\]?)/; +$jobid=$1 || pod2usage "Not a valid job ID $jobid\n"; + +my $pbs_server = `qstat -Bf|grep ^Server`; +chomp($pbs_server); +$pbs_server=~s/^Server: //; +die "No PBS server found. Is it online?\n" unless $pbs_server; + +my $node=`qstat -f $jobid|grep exec_host`; +$node =~/exec_host\s+=\s+(\w+)/; +$node = $1 || "No valid host found. Is $jobid a live job?\n"; +my $cmd; + +if ($method=~/^c/){ + $cmd = 'cat'; +}elsif ($method=~/^ta/){ + $cmd = 'tail'; +}elsif ($method=~/^f/){ + $cmd = 'tail -f'; +}elsif ($method=~/^h/){ + $cmd = 'head'; +}else { + $cmd = $method; +} +if ($method eq 'top'){ + system("ssh $node -t top"); +}else{ + $exec.= " $node $cmd $spool_dir/$jobid.$pbs_server.OU" if !$show_error; + $exec.= " $node $cmd $spool_dir/$jobid.$pbs_server.ER" if $show_error; + system($exec); +} +print "\n#\tQPEEK complete for job $jobid. Host was: $node\n"; + diff --git a/99.scripts/trinity_utils/util/PBS/trinity_kill.pl b/99.scripts/trinity_utils/util/PBS/trinity_kill.pl new file mode 100644 index 0000000..39375c2 --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/trinity_kill.pl @@ -0,0 +1,36 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $me = $ENV{'USER'}; +my @jobs_running_ln = `qstat -u $ENV{'USER'}`; +if (!@jobs_running_ln || scalar(@jobs_running_ln)<1){ + print "No running jobs for user $me\n"; + exit(); +} + +my %jobs_running; +foreach my $job_ln (@jobs_running_ln){ + $job_ln=~/^(\S+)/; + $jobs_running{$1}=1 if $1; +} + +my $dir=$ARGV[0] ? $ARGV[0] : '.'; +if ($dir && -d $dir){ + # read jobnumbers in current directory (and directory passed as variable) and kill them (start at the last job and move to oldest) + if (-s $dir."/jobnumbers.out"){ + my @job_sub = `tac $dir/jobnumbers.out`; + chomp(@job_sub); + foreach my $job (@job_sub){ + system("qdel -W force $job") if $jobs_running{$job}; + # twice to make sure + system("qdel -W force $job >/dev/null 2>/dev/null") if $jobs_running{$job}; + } + unlink($dir."/jobnumbers.out"); + } + else{ + print "No previous jobs found\n"; + } +} + diff --git a/99.scripts/trinity_utils/util/PBS/trinity_kill.sh b/99.scripts/trinity_utils/util/PBS/trinity_kill.sh new file mode 100644 index 0000000..b168853 --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/trinity_kill.sh @@ -0,0 +1,31 @@ +#!/bin/bash + +# much slower than perl version +# read jobnumbers in current directory (and directory passed as variable) and kill them (start at the last job and move to oldest) + +if [ $1 ]; then + if [ -e "$1/jobnumbers.out" ];then + tac "$1/jobnumbers.out" | + while read line + do + echo "$1/jobnumbers.out": Stopping $line + qdel -W force $line + qdel -W force $line >/dev/null 2>/dev/null + done + rm -f "$1/jobnumbers.out" + fi + exit 0 +fi + +if [ -e jobnumbers.out ];then + tac jobnumbers.out | + while read line + do + echo jobnumbers.out: Stopping $line + qdel -W force $line + qdel -W force $line >/dev/null 2>/dev/null + done + rm -f jobnumbers.out +else + echo "No previous jobs found" +fi diff --git a/99.scripts/trinity_utils/util/PBS/trinity_pbs.cont b/99.scripts/trinity_utils/util/PBS/trinity_pbs.cont new file mode 100644 index 0000000..7ea4da9 --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/trinity_pbs.cont @@ -0,0 +1,70 @@ +################################################################################################################################## +########################## ######################################## +########################## Trinity PBS job submission with multi part dependencies ######################################## +########################## ######################################## +################################################################################################################################## +################################################################################################################################## +### Author: Josh Bowden, Alexie Papanicolaou, CSIRO +### Version 1.0 +########### The main control script, that we will run and can be run seperately later if the job fails at intermediate stages ################### +############################################################################################################################################################ + +RUNVAR1=""$HASHBANG" + +echo Killing any running jobs in "$OUTPUTDIR" +trinity_kill.pl "$OUTPUTDIR" 2> /dev/null >/dev/null +echo Checking and submitting any jobs +########################################################################################################################################################### +################ Use the Trinity/Chrysalis checkpoint files to work out what part of job remains ######################################### +################ and queue only those parts that are still required ######################################### + +if [ -s \""$OUTPUTDIR"/Trinity.fasta\" ] ; then + echo 'Trinity seemingly finished: Trinity.fasta present in output directory. If you think this is wrong, delete Trinity.fasta and re-submit' + exit 0 +fi + + +if [ ! -e \""$OUTPUTDIR"/inchworm.K25.L25"$DS"fa.finished\" ] ;then # do the whole analysis + PBS_JOB1=\`qsub "$JOBNAME1".sh\` + PBS_JOB2=\`qsub -W depend=afterok:\$PBS_JOB1 "$JOBNAME2".sh\` + PBS_JOB3=\`qsub -W depend=afternotok:\$PBS_JOB2 "$JOBNAME3".sh\` + PBS_JOB4_2=\`qsub -W depend=afterok:\$PBS_JOB2,afterany:\$PBS_JOB3 "$JOBNAME4".sh\` + echo "$JOBNAME1".sh submitted ; echo \"\$PBS_JOB1\" > jobnumbers.out ; + echo "$JOBNAME2".sh submitted ; echo \"\$PBS_JOB2\" >> jobnumbers.out ; + echo "$JOBNAME3".sh submitted ; echo \"\$PBS_JOB3\" >> jobnumbers.out ; + echo "$JOBNAME4".sh post-job "$JOBNAME2" submitted ; echo \"\$PBS_JOB4_2\" >> jobnumbers.out ; +else # do analysis after inchworm only + if [ ! -e \""$OUTPUTDIR"/chrysalis/GraphFromIwormFasta.finished\" ] ; then # start analysis after inchworm + PBS_JOB2=\`qsub "$JOBNAME2".sh\` + PBS_JOB3=\`qsub -W depend=afternotok:\$PBS_JOB2 "$JOBNAME3".sh\` + PBS_JOB4_2=\`qsub -W depend=afterok:\$PBS_JOB2,afterany:\$PBS_JOB3 "$JOBNAME4".sh\` + echo "$JOBNAME2".sh submitted ; echo \"\$PBS_JOB2\" > jobnumbers.out ; + echo "$JOBNAME3".sh submitted ; echo \"\$PBS_JOB3\" >> jobnumbers.out ; + echo "$JOBNAME4".sh post-job "$JOBNAME2" submitted ; echo \"\$PBS_JOB4_2\" >> jobnumbers.out ; + else + if [ ! -e \""$OUTPUTDIR"/chrysalis/readsToComponents.finished\" ] ;then # start analysis at Chrisyalis ReadsToTranscripts - which is slow due to I/O + PBS_JOB3=\`qsub "$JOBNAME3".sh\` + PBS_JOB4=\`qsub -W depend=afterok:\$PBS_JOB3 "$JOBNAME4".sh\` + echo "$JOBNAME3".sh submitted ; echo \"\$PBS_JOB3\" > jobnumbers.out ; + echo "$JOBNAME4".sh post-job "$JOBNAME3" submitted ; echo \"\$PBS_JOB4\" >> jobnumbers.out ; + else # Run Chrysalis QuantifyGraph and then Butterfly + PBS_JOB4=\`qsub "$JOBNAME4".sh\` + echo "$JOBNAME4".sh submitted ; echo \"\$PBS_JOB4\" > jobnumbers.out ; + fi + fi + echo When JOB ID \"\$PBS_JOB4\" finishes successfully - see output: \"$OUTPUTDIR\"/\"$JOBNAME4\".o\"\$PBS_JOB4\" - either re-run the submit command or use the following command to get all the data into a single file: + echo ' find "$OUTPUTDIR"/chrysalis -name *allProbPaths.fasta -exec cat {} \\; > "$OUTPUTDIR"/Trinity.fasta ' +fi +" + +echo "${RUNVAR1}" | cat -> ""$JOBPREFIX"_run.sh" +chmod 744 ""$JOBPREFIX"_run.sh" +echo "To restart these jobs run either the same command again:" +echo " trinity_pbs.sh " +echo " or the following script: " +echo " "$JOBPREFIX"_run.sh found in the output directory "$OUTPUTDIR" " +echo "To stop these jobs run:" +echo " trinity_kill.pl "$OUTPUTDIR" " +echo "To check progress of these jobs run:" +echo " qstat -u "$USER"" +echo "" diff --git a/99.scripts/trinity_utils/util/PBS/trinity_pbs.header b/99.scripts/trinity_utils/util/PBS/trinity_pbs.header new file mode 100644 index 0000000..d0082ec --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/trinity_pbs.header @@ -0,0 +1,179 @@ +################################################################################################################################## +########################## ######################################## +########################## Trinity PBS job submission with multi part dependencies ######################################## +########################## ######################################## +################################################################################################################################## +### Author: Josh Bowden, Alexie Papanicolaou, CSIRO +### Version 1.0 +### Function definitions and PBS version specific information +################################################################################################################################## + +###################################################################################################################### +### Return usage information if user requests --help -h or another incorrect input flag. +###################################################################################################################### +function F_USAGE { + echo "" + echo " Usage: " + echo "" + echo " To start an analysis: " + echo " trinity_pbs.sh " + echo "" + echo " To stop previously started PBS jobs in the queue: " + echo " trinity_kill.pl OUTPUTDIR " + echo "" + echo " Where:" + echo " = contains specific job details (see below)" + echo " OUTPUTDIR = is path to output data directory" + echo "" + echo " Note: Modify the values in CONFIG_FILE to suit compute cluster and " + echo " input data size and nameing conventions:" + echo "" + echo " contains data and job input details and user defined variables " + echo " and user and cluster specific options:" + echo " UEMAIL, DATADIRECTORY, OUTPUTDIR, JOBPREFIX, ACCOUNT " + echo " STANDARD_JOB_DETAILS" + echo " per job details:" + echo " NCPU, WALLTIME, MEM, MEMDIR, NUMPERARRAYITEM" + echo "" +} + + +###################################################################################################################### +### Function to return the PBS header information, depending on PBS type. +###################################################################################################################### +# NODESCPUS=$(F_GETNODESTRING $1 $MEM $NCPU large $WALLTIME $JOBNAME $ACCOUNT $PBSUSER $MODTRINITY $JOBPREFIX) +# NODESCPUS=$(F_GETNODESTRING $1 $2 $3 $4 $5 $6 $7 $8 $9 ${10} ) +function F_GETNODESTRING { + if [[ $1 = "--pbspro" ]] ; then + echo " +#PBS -l select=1:ncpus="$3":NodeType="$4":mem="$2" +#PBS -l walltime="$5" +#PBS -N "$6" +"$7" +#PBS -j oe +#PBS -m a +#PBS -V +"$8" +"$9" +" + elif [[ $1 = "--pbs" ]] ; then + echo " +#PBS -l nodes=1:ppn="$3" +#PBS -l vmem="$2" +#PBS -l walltime="$5" +#PBS -N "$6" +#PBS -j oe +#PBS -m a +"$8" +#PBS -V + "$9" +" + fi + echo " JOBNAME=$6" + echo " JOBPREFIX=${10}" +} + + +###################################################################################################################### +### Do some checking that files exist etc. +###################################################################################################################### +#DS=$(F_GETNODESTRING $FILENAMEINPUT $DATADIRECTORY $FILENAMESINGLE $FILENAMELEFT $FILENAMERIGHT) +# DS=$(F_GETNODESTRING $1 $2 $3 $4 $5 ) +function SET_DS { + + # add the correct filename extension so we can check if inchworm has been run in part 3 below + if [[ "$1" == *FR* ]] || [[ "$1" == *RF* ]] ; + then + DS="." + else + DS=".DS." + fi + +} + +#AP: i removed this because a) it didn't stop the script from progressing when there was an error, b) it was not handling multiple input files +function F_CHECKFILES { + + # Check that input data file(s) exist. + if [[ "$1" == *--single* ]] ; then + if [ ! -f ""$2""$3"" ] ; then + echo "Input file does not exist: " + echo " "$2""$3"" + exit 1 + fi + else # check that both left and right filenames exist + if [ ! -f ""$2""$4"" ] ; then + echo "Input file does not exist: " + echo " "$2""$4"" + exit 1 + fi + if [ ! -f ""$2""$5"" ] ; then + echo "Input file does not exist: " + echo " "$2""$5"" + exit 1 + fi + fi +} + + +###################################################################################################################### +### Checking that files containing stage specific script data exist +###################################################################################################################### +#F_WRITESCRIPT( scriptname $SOURCENAME ) +#F_WRITESCRIPT( $1 $2 ) +function F_WRITESCRIPT { + if [ -e "$2" ] ; then + source "$2" + else + echo ""$1" requires file \""$2"\" to be present in the current directory" + exit 1 + fi +} + + + +################################################################################################################################## +# This is used to stop all jobs sent to the PBS queue +# requires path to the filename as second argument +# The order of the following tests is important. +################################################################################################################################## +if [[ "$1" = "--rm" ]] || [[ "$1" = "-rm" ]] ; then + if [[ -e $2/jobnumbers.out ]] ; then + cat $2/jobnumbers.out | while read LINE; do + qdel $LINE || { echo "continuing..." ;} + done + #echo " Any parallel " + else + echo " Could not stop jobs as could not open file: " + echo " $2"jobnumbers.out"" + fi + exit 0 +fi + + + +if [[ "$1" = "--help" ]] || [[ "$1" = "-h" ]] || [[ "$1" = "-?" ]]; then + F_USAGE + exit 0 +fi + + +################################################################################################################################## +## User email information required from command line: +## Provide a valid email address so PBS can email user on start and finish of execution of individual parts (not part 4b or 5b though) +################################################################################################################################## + +if [[ ! -e "$1" ]] ; then + echo "input file does not exist: "$1"" + echo "Please provide an input file" + F_USAGE + exit 0 +fi + + + + + + + + diff --git a/99.scripts/trinity_utils/util/PBS/trinity_pbs.p1 b/99.scripts/trinity_utils/util/PBS/trinity_pbs.p1 new file mode 100644 index 0000000..2336b1c --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/trinity_pbs.p1 @@ -0,0 +1,29 @@ +################################################################################################################################## +########################## ######################################## +########################## Trinity PBS job submission with multi part dependencies ######################################## +########################## ######################################## +################################################################################################################################## +### Author: Josh Bowden, Alexie Papanicolaou, CSIRO +### Version 1.0 +### Inchworm P1 script +################################################################################################################################## + +if [[ $MEM_P1 =~ ^([0-9]+) ]]; then + let MEM_BASE="${BASH_REMATCH[1]}" + JM_MEM="$MEM_BASE"G +else + echo No memory given for kmer counter: "$MEM_P1" + exit 1 +fi + +JOBSTRING1=""$HASHBANG" +"$NODESCPUS" + + cd "$OUTPUTDIR" + export OMP_NUM_THREADS="$NCPU_P1" + export KMP_AFFINITY=compact + # this runs Inchworm only + "$STANDARD_JOB_DETAILS" --JM "$JM_MEM" --CPU "$NCPU_P1" --no_run_chrysalis +" +# Write the JOBSTRING1 to a file for later execution +echo "${JOBSTRING1}" | cat -> ""$JOBNAME1".sh" diff --git a/99.scripts/trinity_utils/util/PBS/trinity_pbs.p2 b/99.scripts/trinity_utils/util/PBS/trinity_pbs.p2 new file mode 100644 index 0000000..f208aa6 --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/trinity_pbs.p2 @@ -0,0 +1,43 @@ +################################################################################################################################## +########################## ######################################## +########################## Trinity PBS job submission with multi part dependencies ######################################## +########################## ######################################## +################################################################################################################################## +### Author: Josh Bowden, Alexie Papanicolaou, CSIRO +### Version 1.0 +### Chrysalis P2 script +################################################################################################################################## + +if [[ $MEM_P2 =~ ^([0-9]+) ]]; then + let MEM_BASE="${BASH_REMATCH[1]}" + let ALIGN_MEM=$MEM_BASE-5 + if [ $ALIGN_MEM -le 0 ];then + let ALIGN_MEM=$MEM_BASE-2 + if [ $ALIGN_MEM -le 0 ];then + echo "Memory requested for MEM_P2 is too low. Ask for at least 5 gigabytes" + exit 1 + fi + fi + ALIGN_MEM="$ALIGN_MEM"G +else + echo No memory given: "$MEM_P2" + exit 1 +fi + + +JOBSTRING2=""$HASHBANG" +"$NODESCPUS" + cd "$OUTPUTDIR" + # set stack size to unlimited for Chrysalis (part 1) + ulimit -s unlimited + export OMP_NUM_THREADS="$NCPU_P2" + export KMP_AFFINITY=scatter + # this runs Chrysalis::GraphFromFasta and maybe Chrysalis::GraphFromFasta if there is still walltime + "$STANDARD_JOB_DETAILS" --JM "$ALIGN_MEM" --CPU "$NCPU_P2" --no_run_quantifygraph +" + + +# Write the JOBSTRING2 to a file for later execution +echo "${JOBSTRING2}" | cat -> ""$JOBNAME2".sh" + + diff --git a/99.scripts/trinity_utils/util/PBS/trinity_pbs.p3 b/99.scripts/trinity_utils/util/PBS/trinity_pbs.p3 new file mode 100644 index 0000000..3a5c971 --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/trinity_pbs.p3 @@ -0,0 +1,38 @@ +################################################################################################################################## +########################## ######################################## +########################## Trinity PBS job submission with multi part dependencies ######################################## +########################## ######################################## +################################################################################################################################## +### Author: Josh Bowden, Alexie Papanicolaou, CSIRO +### Version 1.0 +### Chrysalis P3 script (only run if Chrysalis P3 has not completed) +################################################################################################################################## + +if [[ $MEM_P3 =~ ^([0-9]+) ]]; then + let MEM_BASE="${BASH_REMATCH[1]}" + let ALIGN_MEM=$MEM_BASE-5 + if [ $ALIGN_MEM -le 0 ];then + let ALIGN_MEM=$MEM_BASE-2 + if [ $ALIGN_MEM -le 0 ];then + echo "Memory requested for MEM_P3 is too low. Ask for at least 5 gigabytes" + exit 1 + fi + fi + ALIGN_MEM="$ALIGN_MEM"G +else + echo No memory given: "$MEM_P3" + exit 1 +fi + +JOBSTRING3=""$HASHBANG" +"$NODESCPUS" + cd "$OUTPUTDIR" + ulimit -s unlimited + export OMP_NUM_THREADS="$NCPU_P3" + export KMP_AFFINITY=scatter + # this runs Chrysalis::ReadsToTranscripts if it has not completed in the previous step + "$STANDARD_JOB_DETAILS" --JM "$ALIGN_MEM" --CPU "$NCPU_P3" --no_run_quantifygraph +" + +# Write the JOBSTRING3 to a file for later execution +echo "${JOBSTRING3}" | cat -> ""$JOBNAME3".sh" diff --git a/99.scripts/trinity_utils/util/PBS/trinity_pbs.p4a b/99.scripts/trinity_utils/util/PBS/trinity_pbs.p4a new file mode 100644 index 0000000..3fc38e4 --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/trinity_pbs.p4a @@ -0,0 +1,156 @@ +################################################################################################################################## +########################## ######################################## +########################## Trinity PBS job submission with multi part dependencies ######################################## +########################## ######################################## +################################################################################################################################## +### Author: Josh Bowden, Alexie Papanicolaou, CSIRO +### Version 1.0 +### QuantifyGraph and Butterfly p4a Script +################################################################################################################################## +# we will not use array in order to ensure only jobs that have not finished are Submitting. +# this does cause a problem when wanting to kill them manually but best to use kill script +## JOBPREFIX is passed via HASHBANG +JOBSTRING4=""$HASHBANG" +"$NODESCPUS" + JOB_CHRYSALIS="$JOBPREFIX"_p4b + JOB_BUTTERFLY="$JOBPREFIX"_p5b + cd "$OUTPUTDIR" + + rm -f \$JOB_CHRYSALIS.jobnames \$JOB_BUTTERFLY.jobnames + FILENAME=""$OUTPUTDIR"/chrysalis/quantifyGraph_commands" + FILENAMEBFLY=""$OUTPUTDIR"/chrysalis/butterfly_commands" + + if [ ! -e \$FILENAME.pbs ];then + sed -e 's/.*/if [[ -e SEDPLACEHOLDER ]]; then &;fi/' \$FILENAME|sed -r 's/(-e\s)SEDPLACEHOLDER(\s.+-i\s)(\S+).tmp/\1\3.tmp\2\3.tmp/' > \$FILENAME.pbs + split -d -a 3 -l "1000" \$FILENAME.pbs \$FILENAME.pbs. + sleep 5 + fi + if [ ! -e \$FILENAMEBFLY.pbs ];then + sed -e 's/.*/if [ -e SEDPLACEHOLDER ]; then &;fi/' \$FILENAMEBFLY|sed -r 's/(-e\s)SEDPLACEHOLDER(\s.+-C\s)(\S+)/\1\3.out\2\3/' > \$FILENAMEBFLY.pbs + split -d -a 3 -l "1000" \$FILENAMEBFLY.pbs \$FILENAMEBFLY.pbs. + sleep 5 + fi + + FILENAME=\$FILENAME.pbs + FILENAMEBFLY=\$FILENAMEBFLY.pbs + + NUMCMDS=\`ls -l \$FILENAME.??? | wc -l\` + NUMCMDSBFLY=\`ls -l \$FILENAMEBFLY.??? | wc -l\` + let NUMCMDS=\$NUMCMDS-1 + let NUMCMDSBFLY=\$NUMCMDSBFLY-1 + let SUBMITTED_C=0 + let SUBMITTED_B=0 + + for ((JOBID=0;JOBID<=\$NUMCMDS;++JOBID));do + JOB_INDEX_PADDED=\`printf "%03d" \$JOBID\` + MYJOBQ=\""\$FILENAME".\$JOB_INDEX_PADDED\" + MYJOBB=\""\$FILENAMEBFLY".\$JOB_INDEX_PADDED\" + JOB_FILESIZE_Q=\$(stat -c%s \$MYJOBQ) + JOB_FILESIZE_B=\$(stat -c%s \$MYJOBB) + + #if some Q have completed: + if [ -s \"\$MYJOBQ.completed\" ] ; then + JOB_COMPLETED_FILESIZE_Q=\$(stat -c%s \"\$MYJOBQ.completed\") + # if not all have completed then run both Q and B + if [ \"\$JOB_FILESIZE_Q\" -gt \"\$JOB_COMPLETED_FILESIZE_Q\" ] ; then + PBS_JOB4=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_CHRYSALIS.sh\" \` + if [[ ! \$PBS_JOB4 ]]; then + echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_CHRYSALIS.sh\" FAILED. Aborting...\" + exit 255 + fi + let SUBMITTED_C++ + echo \$PBS_JOB4 >> \"\$JOB_CHRYSALIS.jobnames\" + echo \$PBS_JOB4 >> jobnumbers.out ; + PBS_JOB5=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED -W depend=afterok:$PBS_JOB4 \"\$JOB_BUTTERFLY.sh\" \` + if [[ ! \$PBS_JOB5 ]]; then + echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED -W depend=afterok:$PBS_JOB4 \"\$JOB_BUTTERFLY.sh\" FAILED. Aborting...\" + exit 255 + fi + let SUBMITTED_B++ + echo \$PBS_JOB5 >> \"\$JOB_BUTTERFLY.jobnames\" + echo \$PBS_JOB5 >> jobnumbers.out ; + if [ \$(( \$JOBID % 20 )) -eq 0 ] ; then + echo Submitting up to 20 Quantify and/or Butterfly jobs + sleep 3 # be nice + fi + # else all Q have completed; have B completed? + else + # if at least some B have completed + if [ -s \"\$MYJOBB.completed\" ] ; then + JOB_COMPLETED_FILESIZE_B=\$(stat -c%s \"\$MYJOBB.completed\" ) + # if not all, run them with no dependency (Q has completed) + if [ \"\$JOB_FILESIZE_B\" -gt \"\$JOB_COMPLETED_FILESIZE_B\" ] ; then + PBS_JOB5=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_BUTTERFLY.sh\" \` + if [[ ! \$PBS_JOB5 ]]; then + echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_BUTTERFLY.sh\" FAILED. Aborting...\" + exit 255 + fi + let SUBMITTED_B++ + echo \$PBS_JOB5 >> \"\$JOB_BUTTERFLY.jobnames\" + echo \$PBS_JOB5 >> jobnumbers.out ; + if [ \$(( \$JOBID % 20 )) -eq 0 ] ; then + echo Submitting up to 20 Quantify and/or Butterfly jobs + sleep 3 # be nice + fi + fi + # else no Q have completed; run them without dependency + else + PBS_JOB5=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_BUTTERFLY.sh\" \` + if [[ ! \$PBS_JOB5 ]]; then + echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_BUTTERFLY.sh\" FAILED. Aborting...\" + exit 255 + fi + let SUBMITTED_B++ + echo \$PBS_JOB5 >> \"\$JOB_BUTTERFLY.jobnames\" + echo \$PBS_JOB5 >> jobnumbers.out ; + if [ \$(( \$JOBID % 20 )) -eq 0 ] ; then + echo Submitting up to 20 Quantify and/or Butterfly jobs + sleep 3 # be nice + fi + fi + fi + # neither Q (and thus nor B) have ever ran successfully, submit both with a dependency + else + PBS_JOB4=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_CHRYSALIS.sh\" \` + if [[ ! \$PBS_JOB4 ]]; then + echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED \"\$JOB_CHRYSALIS.sh\" FAILED. Aborting...\" + exit 255 + fi + let SUBMITTED_C++ + echo \$PBS_JOB4 >> \"\$JOB_CHRYSALIS.jobnames\" + echo \$PBS_JOB4 >> jobnumbers.out ; + PBS_JOB5=\`qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED -W depend=afterok:\$PBS_JOB4 \"\$JOB_BUTTERFLY.sh\" \` + if [[ ! \$PBS_JOB5 ]]; then + echo \"Submission for qsub -v JOB_INDEX_PADDED=\$JOB_INDEX_PADDED -W depend=afterok:\$PBS_JOB4 \"\$JOB_BUTTERFLY.sh\" FAILED. Aborting...\" + exit 255 + fi + let SUBMITTED_B++ + echo \$PBS_JOB5 >> \"\$JOB_BUTTERFLY.jobnames\" + echo \$PBS_JOB5 >> jobnumbers.out ; + if [ \$(( \$JOBID % 20 )) -eq 0 ] ; then + echo Submitting up to 20 Quantify and/or Butterfly jobs + sleep 3 # be nice + fi + fi + done + + echo Submitted \$SUBMITTED_C Chrysalis and \$SUBMITTED_B Butterfly jobs + + if [[ \$SUBMITTED_B == 0 && \$SUBMITTED_C == 0 ]]; then + echo \"No Trinity jobs need to be submitted \" + if [ -s "$OUTPUTDIR"/Trinity.fasta.complete ]; then + echo \"Trinity RNA-Seq assembly is complete! Result file is present as "$OUTPUTDIR"/Trinity.fasta \" + else + echo \"Proceeding with capturing the output with this command\" + echo ' find "$OUTPUTDIR"/chrysalis -name *allProbPaths.fasta -exec cat {} \\; > "$OUTPUTDIR"/Trinity.fasta ' + find "$OUTPUTDIR"/chrysalis -name *allProbPaths.fasta -exec cat {} \\; > "$OUTPUTDIR"/Trinity.fasta + touch "$OUTPUTDIR"/Trinity.fasta.complete + echo DO: rm -f "$OUTPUTDIR"/bowtie.nameSorted.sam* "$OUTPUTDIR"/both.fa* "$OUTPUTDIR"/inchworm.kmer_count "$OUTPUTDIR"/iworm_* "$OUTPUTDIR"/target* "$OUTPUTDIR"/jellyfish* "$OUTPUTDIR"/scaffolding* "$OUTPUTDIR"/*.finished "$OUTPUTDIR"/mer_counts_* + fi + fi +" + + +###### +##### Write the above script to a file for later execution +echo "${JOBSTRING4}" | cat -> "$JOBNAME4.sh" diff --git a/99.scripts/trinity_utils/util/PBS/trinity_pbs.p4b b/99.scripts/trinity_utils/util/PBS/trinity_pbs.p4b new file mode 100644 index 0000000..7252820 --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/trinity_pbs.p4b @@ -0,0 +1,42 @@ +################################################################################################################################## +########################## ######################################## +########################## Trinity PBS job submission with multi part dependencies ######################################## +########################## ######################################## +################################################################################################################################## +### Author: Josh Bowden, Alexie Papanicolaou, CSIRO +### Version 1.0 +### QuantifyGraph p4b Script +################################################################################################################################## + +JOBSTRING4b=""$HASHBANG" + "$NODESCPUS" + if [[ ! \$JOB_INDEX_PADDED ]];then + echo \"Error: not a proper submission\" + exit 255 + fi + echo \"Processing quantifyGraph_commands index \$JOB_INDEX_PADDED \" + cd "$OUTPUTDIR" + export OMP_NUM_THREADS=1 + COREFILENAME=""$OUTPUTDIR"/chrysalis/quantifyGraph_commands.pbs" + MYJOBQ=\$COREFILENAME.\$JOB_INDEX_PADDED + JOB_FILESIZE=\$(stat -c%s \"$MYJOBQ\") + if [ -s \"\$MYJOBQ.completed\" ] ; then + JOB_COMPLETED_FILESIZE=\$(stat -c%s \"\$MYJOBQ.completed\") + if [ \"\$JOB_FILESIZE\" != \"\$JOB_COMPLETED_FILESIZE\" ] ; then + trap \" echo \\\"Please check \$MYJOBQ Chrysalis QuantifyGraph processes had enough walltime.\\\"; exit 255 \" INT TERM + "$TRINITYPATH"/trinity-plugins/parafly/bin/ParaFly -CPU "$NCPU_P4" -v -failed_cmds \$MYJOBQ.failed -c \$MYJOBQ + trap - INT TERM EXIT + fi + else + trap \" echo \\\"Please check \$MYJOBQ Chrysalis QuantifyGraph processes had enough walltime.\\\"; exit 255 \" INT TERM + "$TRINITYPATH"/trinity-plugins/parafly/bin/ParaFly -CPU "$NCPU_P4" -v -failed_cmds \$MYJOBQ.failed -c \$MYJOBQ + trap - INT TERM EXIT + fi + + sleep 30 # IO friendship for following butterfly job - sometimes butterfly fails to find output if io is overwhelmed + exit + +" +# Write the above script to a file for later execution +echo "${JOBSTRING4b}" | cat -> "$JOBPREFIX"_p4b.sh + diff --git a/99.scripts/trinity_utils/util/PBS/trinity_pbs.p5b b/99.scripts/trinity_utils/util/PBS/trinity_pbs.p5b new file mode 100644 index 0000000..cd280dc --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/trinity_pbs.p5b @@ -0,0 +1,41 @@ +################################################################################################################################## +########################## ######################################## +########################## Trinity PBS job submission with multi part dependencies ######################################## +########################## ######################################## +################################################################################################################################## +### Author: Josh Bowden, Alexie Papanicolaou, CSIRO +### Version 1.0 +### Butterfly p5b Script +################################################################################################################################## + +JOBSTRING5b=""$HASHBANG" + "$NODESCPUS" + if [[ ! \$JOB_INDEX_PADDED ]];then + echo \"Error: not a proper submission\" + exit 255 + fi + echo \"Processing butterfly_commands index \$JOB_INDEX_PADDED \" + cd "$OUTPUTDIR" + export OMP_NUM_THREADS=1 + COREFILENAME=""$OUTPUTDIR"/chrysalis/butterfly_commands.pbs" + MYJOBB=\$COREFILENAME.\$JOB_INDEX_PADDED + JOB_FILESIZE=\$(stat -c%s \"\$MYJOBB\") + if [ -s \"\$MYJOBB.completed\" ] ; then + JOB_COMPLETED_FILESIZE=\$(stat -c%s \"\$MYJOBB.completed\") + if [ \"$JOB_FILESIZE\" != \"\$JOB_COMPLETED_FILESIZE\" ] ; then + trap \" echo \\\"Please check \$MYJOBB Butterfly processes had enough walltime.\\\"; exit 255 \" INT TERM + "$TRINITYPATH"/trinity-plugins/parafly/bin/ParaFly -CPU "$NCPU_P5" -v -failed_cmds \$MYJOBB.failed -c \$MYJOBB + trap - INT TERM EXIT + fi + else + trap \" echo \\\"Please check \$MYJOBB Butterfly processes had enough walltime.\\\"; exit 255 \" INT TERM + "$TRINITYPATH"/trinity-plugins/parafly/bin/ParaFly -CPU "$NCPU_P5" -v -failed_cmds \$MYJOBB.failed -c \$MYJOBB + trap - INT TERM EXIT + fi +exit + + +" +# Write the above script to a file for later execution +echo "${JOBSTRING5b}" | cat -> "$JOBPREFIX"_p5b.sh + diff --git a/99.scripts/trinity_utils/util/PBS/trinity_pbs.sh b/99.scripts/trinity_utils/util/PBS/trinity_pbs.sh new file mode 100644 index 0000000..0195302 --- /dev/null +++ b/99.scripts/trinity_utils/util/PBS/trinity_pbs.sh @@ -0,0 +1,291 @@ +#!/bin/bash +set -e # turn on exit on error +################################################################################################################################## +########################## ######################################## +########################## Trinity PBS job submission with multi part dependencies ######################################## +########################## ######################################## +################################################################################################################################## +### Author: Josh Bowden, Alexie Papanicolaou, CSIRO +### Version 1.0 +### +### +### Script to split the Trinity workflow into multiple stages so as to efficiently request +### and use appropriate resources (walltime and number of cores) on a computer cluster / supercomputer. +### Currently creates scripts for PBS Torque or PBSpro +### +### trinity_pbs script install instructions: +### 1. Copy all trinity_pbs.* files into a directory (we will call it "TRINITY_PBS_DIR"). +### 2. Add TRINITY_PBS_DIR to the PATH i.e. export or set PATH=TRINITY_PBS_DIR:$PATH (perhaps export PATH in .bashrc file) +### 3. Change the "TRINITYPBSPATH" variable found below to point to the directory also. i.e. TRINITYPBSPATH=TRINITY_PBS_DIR +### 4. Set MEMDIRIN to name of a node-local filesystem so a network drive is not needed unecesarily for Scripts 4b and 5b +### 5. Set MODTRINITY to any modules that need to be loaded so Trinity.pl can be run. +### 6. Set TRINITYPATH to the path to Trinity.pl executable +### 7. Set PBSTYPE to --pbspro or --pbs, dependent on the system present. +### That should be all that is needed from an admin perspective (besides making scripts accessible and exectable for users) +### +### Users need make a copy of TRINITY.CONFIG.template and then modify variables in it. See TRINITY.CONFIG.template for further details. +### +### The current script does the following. +### Part 1. Reads data from TRINITY.CONFIG and creates the input directory, data file names, output data directory and Trinity.pl command line +### User inputs from TRINITY.CONFIG file : +### JOBPREFIX A string of less than 11 characters long. PBS will use this as a jobname prefix. +### DATADIRECTORY Where input data exists +### OUTPUTDIR Where user wants output data to go - requires a lot of space even for small datatsets +### STANDARD_JOB_DETAILS the Trinity.pl command line +### ACCOUNT Account details of user (if required by PBS system being used) +### +### Part 2. Writes scripts to run Trinity.pl in 6 stages: +### 3 intial (Inchworm, and 2 x Chrysalis stages: Chrysalis::GraphFromFasta and Chrysalis::ReadsFromTranscripts) +### 2 parallel stages (Chrysalis::QuantifyGraph and Butterfly) which are executed in parallel. +### 1 collection of results as Trinity.Fasta. +### +### Information input from command line filename for stage 'x' : +### WALLTIME_Px Amount of time stage requires +### MEM_Px The amount of memory the stage requires +### NCPU_Px The number of CPUs the stage may use +### PBSNODETYPE _Px The PBS (for --pbspro only) queue name +### NUMPERARRAYITEM_Px The number of massively parallel jobs in each parallel satge. +### +### Part 3. Runs scripts dependant upon what stage has been detected as completed, using PBS job dependencies +### +### Command line usage: +### To start (or re-start) an analysis: +### >trinity_pbs.sh TRINITY.CONFIG.template +### To stop previously started PBS jobs on the queue: +### >trinity_kill.pl OUTPUTDIR +### Where: +### TRINITY.CONFIG.template = user specific job details +### OUTPUTDIR = is path to output data directory +### +### Output job script submission files. These are saved in the output directory (OUTPUTDIR) and can be modified/re-run if any job fails. +### *_run.sh Runs all the following scripts - with job dependencies and only the jobs that still need to be run. +### *_p1.sh Runs Inchworm stage. Does not scale well past a single socket. Only request at most the number of cores on a single CPU. +### *_p2.sh Runs Chrysalis::GrapghFromFasta clustering of Inchworm output. Should scale to number of cores on node +### *_p3.sh Runs Chrysalis::ReadsToTranscripts. I/O limited. Try to use local filesystem (not implemented) +### *_p4a.sh Creates jobs to run Chrysalis::QuantifyGraph and Butterfly parallel tasks +### *_p4b.sh QuantifyGraph job. "NUMPERARRAYITEM" tasks from the file /chrysalis/quantifyGraph_commands are run for each job +### *_p5b.sh Butterfly job. "NUMPERARRAYITEM" tasks from the file /chrysalis/butterfly_commands are run for each job +### to start off at last completed stage. At present leaves all data on temporary area of shared network drive and +### copies Trinity.fatsa to home directory (with specific job prefix in filename). +### * = $JOBPREFIX. $JOBPREFIX should not be > 10 characters long + + +################################################################################################################################## +####################### SET TRINITY INSTALLATION PATH (where Trinity.pl resides ################################################# +################################################################################################################################## +# We need the path even if loaded using module +TRINITYPATH="/home/pap056/software/trinity_2013_08_14/" +# If you are loading using module, you can make the next variable blank +NEWPATH= +NEWPATH="export PATH=$PATH:$TRINITYPATH" + +################################################################################################################################## +######## Set TRINITYPBSPATH to the directory where trinity_pbs.sh scripts are installed +################################################################################################################################## +TRINITYPBSPATH=`dirname "$0"`; # set to location of this script + +################################################################################################################################## +######## Set cluster specific name for compute node local filesystem +################################################################################################################################## +MEMDIRIN="\$TMPDIR" # available on Barrine + + +################################################################################################################################## +######## Set system specific PATHS and load system specific modules (if available) +################################################################################################################################## +# Example for for Barrine: +PBSTYPE="--pbspro" +# Here we ensure that Java 1.6 is used and Java 1.7 is removed (Butterfly dependency) +MODTRINITY=" + module load mpt/2.00 perl/5.15.8 bowtie/12.7 jellyfish/1.1.5 samtools/1.18 java/1.6.0_22-sun; + module rm java/1.7.0_02 +" + +## That should be all the admin modifications needed. + +# Append Trinity path data to env variables loaded by every script +MODTRINITY=" + $NEWPATH; + export TRINITYPATH="/home/pap056/software/trinity_2013_08_14"; + $MODTRINITY +" +################################################################################################################################## +################################################################################################################################## +################################################################################################################################## + +################################################################################################################################## +######### Load files that contains functions +################################################################################################################################## +# Modify function F_GETNODESTRING in file trinity_pbs.header so that a correct PBS header is returned to suit your PBS cluster +if [ -e "$TRINITYPBSPATH"/trinity_pbs.header ] ; then + source "$TRINITYPBSPATH"/trinity_pbs.header +else + echo "$1 requires file \"trinity_pbs.header\" to be present in: " + echo "$TRINITYPBSPATH" + exit 1 +fi + +################################################################################################################################## +######### Load input config file +################################################################################################################################## +if [ -e "$1" ] ; then + source "$1" +else + echo "Error: Input file does not exist: "$1" " + exit 1 +fi + + +################################################################################################################################## +## Common variables to PBS and PBSpro +## and other needed variables that a user should not need to modify +################################################################################################################################## +if [ $UEMAIL ]; then PBSUSER="#PBS -M "$UEMAIL"" ; fi +HASHBANG="#!/bin/bash" + +################################################################################################################################## +## PBS torque and PBSpro have some differences. +## Organise these here and also check further on (line 184) and change NODETYPE to match cluster system +## MODTRINITY will also be different on different clusters - it sets up the paths to the required executables +################################################################################################################################## +if [[ "$PBSTYPE" = "--pbspro" ]] ; then + JOBARRAY="-J" + JOBARRAY_ID="\$PBS_ARRAY_INDEX" + AFTEROKARRAY="afterok" +elif [[ "$PBSTYPE" = "--pbs" ]] ; then + ## PBS torque: + JOBARRAY="-t" + JOBARRAY_ID="\$PBS_ARRAYID" + AFTEROKARRAY="afterokarray" +else # no paramaters present + F_USAGE + exit 0 +fi + +############################################################################################################################################## +########## Part 1: Set up file names for input directory and for output data dir and Trinity.pl command line #################### +############################################################################################################################################## + echo "" + ## Ensure JOBPREFIX is not greatr than 11 characters as PBS-pro can not handle > 15 characters for total job name length + echo "submitting trinity jobs with prefix: " + echo " $JOBPREFIX" + + ###### Set input data directory - $DATADIR is CSIRO specific + echo "Input directory: " + echo " $DATADIRECTORY" + + ###### Set output data directory (OUTPUTDIR) + echo "Output directory: Scripts and output data will be written to:" + echo " $OUTPUTDIR" + mkdir -p "$OUTPUTDIR" + cd "$OUTPUTDIR" + + + ### Modify STANDARD_JOB_DETAILS for analysis specific input to Trinity.pl + echo "The following trinity command line will be run:" + echo "$STANDARD_JOB_DETAILS" + echo "" + if [[ "$1" = "--pbs" ]] ; then + echo " Use: \"pbs_check.pl -t PBS_JOBID\" " + echo " To view stdout and stderr from each separate job while they are running" + fi +########################################################################################################################################################### +### Do some checking that files exist etc. (User should not modify) +### This sets the $DS variable + +SET_DS "$FILENAMEINPUT" + +########################################################################################################################################################### +################################# ################################### +################################# Part 2: Create the shell scripts to be run via the PBS batch system ################################### +################################# Users should modify WALLTIME and MEM dependent upon dataset size ################################### +################################# and NCPU to appropriate value for compute node cpu resources ################################### +################################# Check: "Trinity RNA-seq Assembler Performance Optimisation" (Henschel 2012) ################################### +################################# for current best practice. ################################### +################################# N.B. On busy clusters it may be best not to try to request ################################### +################################# a full nodes resources. i.e. If 8 cores per node are present, ################################### +################################# only request half of these. ################################### +########################################################################################################################################################### + +########################################################################################################################################################### +############################## Script 1: Write script to run Inchworm ############################################ +############################# MEM should equal JFMEM, which is the amount of memory requested for Jellyfish ############################################ + +JOBNAME1="$JOBPREFIX"_p1 +NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P1" "$NCPU_P1" "$PBSNODETYPE_P1" "$WALLTIME_P1" "$JOBNAME1" "$ACCOUNT" "$PBSUSER" "$MODTRINITY" "$JOBPREFIX") + +F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p1" + +############################################################################################################################################################# +############################## Script 2: Chrysalis::GraphFromFasta ############################################# +############################## This script has a dependency on part 1 completion without error. ############################################# +# It would be good to force an exit(0) before ReadsToTranscripts after checkpoint file /chrysalis/GraphFromIwormFasta.finished is +# written (i.e. add --no_run_readstotrans to Trinity and pass through to Chrysalis) as the script 3 can be started directly after. +# This may be less of an issue with the new (fast) version of Trinity::GraphFromFasta (since version 2012-06-08). + + +JOBNAME2="$JOBPREFIX"_p2 +NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P2" "$NCPU_P2" "$PBSNODETYPE_P2" "$WALLTIME_P2" "$JOBNAME2" "$ACCOUNT" "$PBSUSER" "$MODTRINITY" "$JOBPREFIX") + +F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p2" + +########################################################################################################################################################### +############################## Script 3: Script to run Chrysalis::ReadsToTranscripts ################### +############################## ReadsToTranscripts can be slow due to reads from disk, ################### +############################## This script has a dependency on part 2 completion with error. ################### +############################## Section script is skipped if enough time was given in Part 2. ################### + + +JOBNAME3="$JOBPREFIX"_p3 +NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P3" "$NCPU_P3" "$PBSNODETYPE_P3" "$WALLTIME_P3" "$JOBNAME3" "$ACCOUNT" "$PBSUSER" "$MODTRINITY" "$JOBPREFIX") + +F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p3" + + +########################################################################################################################################################## +############################## Script 4a: Write script to call the Chrysalis QuantifyGraph array job ################## +############################## N.B. SLOTLIMIT="%x" indicates 'slot limit' i.e. the number of concurrent jobs to execute in an array ################## +############################## (Not available in PBSpro ) ################## +#### NB Disabling emails for arrays + +#SLOTLIMIT="%64" # available for PBS Torque +JOBNAME4="$JOBPREFIX"_p4a +NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" 1gb 1 "$PBSNODETYPE_P4" 00:30:00 "$JOBNAME4" "$ACCOUNT" "$PBSUSER" "$MODTRINITY" "$JOBPREFIX" ) + +F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p4a" + +########################################################################################################################################################### +############################## Script Array part 4b: Write script to be run as an array Job . Runs Chrysalis::QuantifyGraph ################## +############################## This scipt has a dependency on part 4a being run. If an array component fails it will email user. ################## +############################## Files named quantifyGraph_commands_X are written with subset of total commands (X is the array ID from the PBS system) ## + + +JOBNAME4B="$JOBPREFIX"_p4b +NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P4" "$NCPU_P4" "$PBSNODETYPE_P4" "$WALLTIME_P4" "$JOBNAME4B" "$ACCOUNT" " " "$MODTRINITY" "$JOBPREFIX") + +F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p4b" + + +########################################################################################################################################################### +############################## Script Array 5b: Write script to run Butterfly Array Job ################## +############################## This script has a dependency on part 4a 4b being run. ################## +############################## Files named butterfly_commands_X are written with subset of total commands (X is the array ID from the PBS system) ### + +JOBNAME5B="$JOBPREFIX"_p5b +NODESCPUS=$(F_GETNODESTRING "$PBSTYPE" "$MEM_P5" "$NCPU_P5" "$PBSNODETYPE_P5" "$WALLTIME_P5" "$JOBNAME5B" "$ACCOUNT" " " "$MODTRINITY" "$JOBPREFIX") + +F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.p5b" + +############################################################################################################################################################ +############### ################### +############### Part 3: Write main control script that executes scripts that were created above. ################### +############### ################### +############################################################################################################################################################ +F_WRITESCRIPT "$0" ""$TRINITYPBSPATH"/trinity_pbs.cont" + +############### Run the script written in Part 3 +############### Checks to see what is current stage of calculation and executes scripts created in above code ################### +bash ""$JOBPREFIX"_run.sh" + +exit 0 + diff --git a/99.scripts/trinity_utils/util/R/expression_analysis_lib.R b/99.scripts/trinity_utils/util/R/expression_analysis_lib.R new file mode 100644 index 0000000..9732adc --- /dev/null +++ b/99.scripts/trinity_utils/util/R/expression_analysis_lib.R @@ -0,0 +1,118 @@ + +plot_log2_fpkm_dist = function (fpkm_files) { + + num_files = length(fpkm_files); + + data_list = list(); + + max_y = 0 + xlim = c(0,0) + + for (i in 1:num_files) { + + file = fpkm_files[i] + data = read.table(file, header=F) + data = data[,6]; + data = log2(data+1) + + den = density(data); + + data_list[[i]] = den + + y = max(den$y); + if (y > max_y) { + max_y = y; + } + + x = min(den$x) + if (x < xlim[1]) { + xlim[2] = x + } + + x = max(den$x); + if (x > xlim[2]) { + xlim[2] = x + } + + + } + + colors = rainbow(num_files); + + for (i in 1:num_files) { + + if (i == 1) { + + plot(data_list[[1]], col=colors[1], xlim=xlim, ylim=c(0,max_y), xlab="log2(fpkm+1)") + } + else { + points(data_list[[i]], col=colors[i], type='l') + } + } + + return; +} + + + + + +plot_expressed_gene_counts = function(fpkm_file, + title="expressed transcript counts vs. min fpkm", + fpkm_range=seq(0,10,0.2), + total=0, + outfile="count_summary.txt") { + + data = read.table(fpkm_file, header=T, row.names=1); + data = data[,5] + + + counts_expressed = c(); + counts_not_expressed = c(); + + print_not_expressed_flag = 1; + if (total == 0) { + total = length(data); + print_not_expressed_flag = 0; + } + + count_expressed = total; + for (i in fpkm_range) { + + if (i > 0) { + count_expressed = sum(data>=i) + } + count_not_expressed = total - count_expressed; + + counts_expressed[length(counts_expressed)+1] = count_expressed; + counts_not_expressed[length(counts_not_expressed)+1] = count_not_expressed; + + } + + + orig_settings = par(mfrow=c(1,2)) + + plot(fpkm_range, counts_expressed, type='o', col='black', xlab="min(fpkm)", + main=title, ylab="count of transcripts", ylim=c(0,max(counts_expressed, counts_not_expressed))) + + if (print_not_expressed_flag) { + points(fpkm_range, counts_not_expressed, type='o', col='blue') + legend('topright', c('expressed', 'not expressed'), col=c('black', 'blue'), pch=15); + } + else { + legend('bottomright', c('expressed transcripts'), col=c('black'), pch=15); + } + + data_table = data.frame(fpkm_range=fpkm_range, expressed=counts_expressed, not_expressed=counts_not_expressed); + + write.table(data_table, file=outfile, quote=F, sep='\t', row.names=F); + + + ## make density plot for non-zero FPKM values + data = data[data>0] + plot(density(log2(data)), xlab="log2(fpkm)") + + + par(orig_settings) + +} diff --git a/99.scripts/trinity_utils/util/R/get_Poisson_conf_intervals.R b/99.scripts/trinity_utils/util/R/get_Poisson_conf_intervals.R new file mode 100644 index 0000000..64f1d44 --- /dev/null +++ b/99.scripts/trinity_utils/util/R/get_Poisson_conf_intervals.R @@ -0,0 +1,102 @@ + +get_Poisson_conf_intervals = function(seq_range, quantile_vec = c(0.05,0.95), plot=T) { + + num_rows = length(seq_range) + num_cols = length(quantile_vec) + m = matrix(nrow=num_rows, ncol=num_cols) + + colnames(m) = quantile_vec + rownames(m) = seq_range + + row_count = 0 + for (i in seq_range) { + q = qpois(quantile_vec, i) + row_count = row_count + 1 + m[row_count,] = q + + } + + results = list() + results$mat = m + + if (plot) { + percents = plot_conf_intervals(m) + + results$pct = percents + } + + return(results) + +} + + +get_NB_conf_intervals = function(seq_range, dispersion, quantile_vec = c(0.05,0.95), plot=T) { + + size = 1/dispersion #according to the mu-definition of size in nbinom of R, where var = mean + (1/size)mean^2 + + num_rows = length(seq_range) + num_cols = length(quantile_vec) + m = matrix(nrow=num_rows, ncol=num_cols) + + colnames(m) = quantile_vec + rownames(m) = seq_range + + row_count = 0 + for (i in seq_range) { + q = qnbinom(quantile_vec, mu=i, size=size) + row_count = row_count + 1 + m[row_count,] = q + + } + + results = list() + results$mat = m + + + if (plot) { + percents = plot_conf_intervals(m) + + results$pct = percents + } + + return(results) + +} + + + + + + +plot_conf_intervals = function(m) { + + par(mfrow=c(1,2)) + + c_names = colnames(m) + r_vals = as.numeric(rownames(m)) + + max_val = max(m) + + plot(r_vals, r_vals, xlab="known read counts", ylab="Poisson read counts dist", t='l') + + # plot the confidence levels + line_colors = rainbow(length(c_names)) + for(i in 1:length(c_names)) { + points(r_vals,m[,i], t='l', col=line_colors[i]) + } + + # plot the max percentage of value for 95% conf level. + percents=c() + for (i in 1:length(r_vals)) { + + max_delta = max(abs(m[i,]-r_vals[i])) + percent = max_delta/r_vals[i]*100 + + percents[i]=percent; + } + + plot(r_vals, percents, ylim=c(0,100), xlab="read counts", ylab="percent of value for 95% conf interval") + + percents + +} diff --git a/99.scripts/trinity_utils/util/TrinityStats.pl b/99.scripts/trinity_utils/util/TrinityStats.pl new file mode 100644 index 0000000..79ce2b8 --- /dev/null +++ b/99.scripts/trinity_utils/util/TrinityStats.pl @@ -0,0 +1,160 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../PerlLib"); +use Fasta_reader; +use BHStats; + +my $usage = "\n\nusage: $0 transcripts.fasta\n\n"; + +my $fasta_file = $ARGV[0] or die $usage; + +main: { + + my $fasta_reader = new Fasta_reader($fasta_file); + + my @all_seq_lengths; + + my $number_transcripts = 0; + + my $num_GC = 0; + + my %component_to_longest_isoform; + + my $tot_seq_len = 0; + + my $missing_gene_ids_flag = 0; + + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + + $number_transcripts++; + my $comp_id = $acc; + + + if (! $missing_gene_ids_flag) { + + if ($acc =~ /^(.*c\d+_g\d+)/) { + $comp_id = $1; + } + elsif ($acc =~ /^(.*comp\d+_c\d+)/) { + $comp_id = $1; + } + else { + print STDERR "Error, cannot decipher gene identifier from acc: $acc"; + $missing_gene_ids_flag = 1; + } + } + + my $sequence = $seq_obj->get_sequence(); + + my $seq_len = length($sequence); + + $tot_seq_len += $seq_len; + + if ( (! exists $component_to_longest_isoform{$comp_id}) + || + $component_to_longest_isoform{$comp_id} < $seq_len) { + + $component_to_longest_isoform{$comp_id} = $seq_len; + } + + push (@all_seq_lengths, $seq_len); + + while ($sequence =~ /[gc]/ig) { + $num_GC++; + } + + } + + print "\n\n"; + print "################################\n"; + print "## Counts of transcripts, etc.\n"; + print "################################\n"; + + print "Total trinity 'genes':\t" . scalar(keys %component_to_longest_isoform) . "\n"; + print "Total trinity transcripts:\t" . $number_transcripts . "\n"; + + + my $pct_gc = sprintf("%.2f", $num_GC / $tot_seq_len * 100); + + + print "Percent GC: $pct_gc\n\n"; + + print "########################################\n"; + print "Stats based on ALL transcript contigs:\n"; + print "########################################\n\n"; + + &report_stats(@all_seq_lengths); + print "\n\n"; + + if ($missing_gene_ids_flag) { + print " - note: not reporting gene-based longest isoform info since couldn't parse Trinity accession info.\n"; + } + else { + print "#####################################################\n"; + print "## Stats based on ONLY LONGEST ISOFORM per 'GENE':\n"; + print "#####################################################\n\n"; + + &report_stats(values %component_to_longest_isoform); + + print "\n\n\n"; + } + + + exit(0); +} + +#### +sub report_stats { + my (@seq_lengths) = @_; + + @seq_lengths = reverse sort {$a<=>$b} @seq_lengths; + + my $cum_seq_len = 0; + foreach my $len (@seq_lengths) { + $cum_seq_len += $len; + } + + for (my $i = 10; $i <= 50; $i += 10) { + my $cum_len_needed = $cum_seq_len * $i/100; + my $N_val = &get_contigNvalue($cum_len_needed, \@seq_lengths); + print "\tContig N$i: $N_val\n"; + } + print "\n"; + + + my $median_len = &BHStats::median(@seq_lengths); + print "\tMedian contig length: $median_len\n"; + + my $avg_len = sprintf("%.2f", &BHStats::avg(@seq_lengths)); + print "\tAverage contig: $avg_len\n"; + + + + print "\tTotal assembled bases: $cum_seq_len\n"; + + return; +} + + + +sub get_contigNvalue { + my ($cum_len_needed, $seq_lengths_aref) = @_; + + my $partial_sum_len = 0; + foreach my $len (@$seq_lengths_aref) { + $partial_sum_len += $len; + + if ($partial_sum_len >= $cum_len_needed) { + return($len); + } + } + + + return -1; # shouldn't happen. +} diff --git a/99.scripts/trinity_utils/util/abundance_estimates_to_matrix.pl b/99.scripts/trinity_utils/util/abundance_estimates_to_matrix.pl new file mode 100644 index 0000000..3598d30 --- /dev/null +++ b/99.scripts/trinity_utils/util/abundance_estimates_to_matrix.pl @@ -0,0 +1,405 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use File::Basename; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use Carp; + +my $usage = <<__EOUSAGE__; + +#################################################################################### +# +# Usage: $0 --est_method sample1.results sample2.results ... +# +# or $0 --est_method --quant_files file.listing_target_files.txt +# +# Note, if only a single input file is given, it's expected to contain the paths to all the target abundance estimation files. +# +# Required: +# +# --est_method RSEM|eXpress|kallisto|salmon (needs to know what format to expect) +# +# --gene_trans_map the gene-to-transcript mapping file. (if you don't want gene estimates, indicate 'none'. +# +# +# Options: +# +# --cross_sample_norm TMM|UpperQuartile|none (default: TMM) +# +# --name_sample_by_basedir name sample column by dirname instead of filename +# --basedir_index default(-2) +# +# --out_prefix default: value for --est_method +# +# --quant_files file containing a list of all the target files. +# +###################################################################################### + + +__EOUSAGE__ + + ; + + +my $help_flag; +my $est_method; +my $val_type; +my $cross_sample_norm = "TMM"; +my $name_sample_by_basedir = 0; +my $out_prefix; +my $basedir_index = -2; +my $quant_files = ""; +my $gene_trans_map_file; + +&GetOptions('help|h' => \$help_flag, + 'est_method=s' => \$est_method, + + 'cross_sample_norm=s' => \$cross_sample_norm, + 'name_sample_by_basedir' => \$name_sample_by_basedir, + 'out_prefix=s' => \$out_prefix, + 'basedir_index=i' => \$basedir_index, + + 'quant_files=s' => \$quant_files, + 'gene_trans_map=s' => \$gene_trans_map_file, + ); + + + +if ($help_flag) { die $usage; } + +unless ($est_method && (@ARGV || $quant_files)) { + die $usage; +} + +unless ($gene_trans_map_file) { + die "Error, specify gene-to-trans map file via: --gene_trans_map, or indicate 'none' if you dont want gene estimates"; +} + +unless ($est_method =~ /^(RSEM|eXpress|kallisto|salmon)/i) { + die "Error, dont recognize --est_method $est_method "; +} +unless ($cross_sample_norm =~ /^(TMM|UpperQuartile|none)$/i) { + die "Error, dont recognize --cross_sample_norm $cross_sample_norm "; +} + +unless ($out_prefix) { + $out_prefix = $est_method; +} + +my @files; + +if ($quant_files) { + # allow for a file listing the various files. + @files = `cat $quant_files`; + chomp @files; +} +elsif (@ARGV) { + @files = @ARGV; +} +else { + die $usage; +} + + + +=data_formats + +## RSEM: + +0 transcript_id +1 gene_id +2 length +3 effective_length +4 expected_count +5 TPM +6 FPKM +7 IsoPct + + +## eXpress v1.5: + +1 target_id +2 length +3 eff_length +4 tot_counts +5 uniq_counts +6 est_counts +7 eff_counts +8 ambig_distr_alpha +9 ambig_distr_beta +10 fpkm +11 fpkm_conf_low +12 fpkm_conf_high +13 solvable +14 tpm + + +## kallisto: +0 target_id +1 length +2 eff_length +3 est_counts +4 tpm + + +## salmon: +0 Name +1 Length +2 EffectiveLength +3 TPM +4 NumReads + +=cut + + ; + +my ($acc_field, $counts_field, $fpkm_field, $tpm_field); + +if ($est_method =~ /^rsem$/i) { + $acc_field = 0; + $counts_field = "expected_count"; + $fpkm_field = "FPKM"; + $tpm_field = "TPM"; +} +elsif ($est_method =~ /^express$/i) { # as of v1.5 + $acc_field = "target_id"; + $counts_field = "eff_counts"; + $fpkm_field = "fpkm"; + $tpm_field = "tpm"; +} +elsif ($est_method =~ /^kallisto$/i) { + $acc_field = "target_id"; + $counts_field = "est_counts"; + $fpkm_field = "tpm"; + $tpm_field = "tpm"; +} +elsif ($est_method =~ /^salmon/) { + $acc_field = "Name"; + $counts_field = "NumReads"; + $fpkm_field = "TPM"; + $tpm_field = "TPM"; +} +else { + die "Error, dont recognize --est_method [$est_method] "; +} + +main: { + + my %data; + + my %sum_sample_counts; + + foreach my $file (@files) { + unless ($file =~ /\w/) { next; } # empty line in quant files listing causes trouble + print STDERR "-reading file: $file\n"; + open (my $fh, $file) or die "Error, cannot open file $file"; + my $header = <$fh>; + chomp $header; + my %fields = &parse_field_positions($header); + #use Data::Dumper; print STDERR Dumper(\%fields); + while (<$fh>) { + chomp; + + my @x = split(/\t/); + my $acc = $x[ $fields{$acc_field} ]; + my $count = $x[ $fields{$counts_field} ]; + my $fpkm = $x[ $fields{$fpkm_field} ]; + my $tpm = $x[ $fields{$tpm_field} ]; + + $data{$acc}->{$file}->{count} = $count; + $data{$acc}->{$file}->{FPKM} = $fpkm; + $data{$acc}->{$file}->{TPM} = $tpm; + + # capture sample total counts + $sum_sample_counts{$file} += $count; + } + close $fh; + } + + my %column_header_to_filename; + my @filenames = @files; + foreach my $file (@filenames) { + my $column_header; + if ($name_sample_by_basedir) { + my @path = split(m|/|, $file); + $column_header = $path[$basedir_index]; + } + else { + $column_header = basename($file); + } + + $column_header =~ s/\.(genes|isoforms)\.results$//; # in case of rsem + + $column_header_to_filename{$column_header} = $file; + $file = $column_header; # update the @filenames + } + print STDERR "\n\n* Outputting combined matrix.\n\n"; + + my $counts_matrix_file = "$out_prefix.isoform.counts.matrix"; + my $TPM_matrix_file = "$out_prefix.isoform.TPM.not_cross_norm"; + open (my $ofh_counts, ">$counts_matrix_file") or die "Error, cannot write file $counts_matrix_file"; + open (my $ofh_TPM, ">$TPM_matrix_file") or die "Error, cannot write file $TPM_matrix_file"; + + { # check to see if they're unique + my %filename_map = map { + $_ => 1 } @filenames; + if (scalar keys %filename_map != scalar @filenames) { + die "Error, the column headings: @filenames are not unique. Should you consider using the --name_sample_by_basedir parameter?"; + } + } + + + # clean up matrix headers + #foreach my $file (@filenames) { + # also, get rid of the part of the filename that RSEM adds + # $file =~ s/\.(genes|isoforms)\.results$//; + #} + + + print $ofh_counts join("\t", "", @filenames) . "\n"; + print $ofh_TPM join("\t", "", @filenames) . "\n"; + + foreach my $acc (keys %data) { + + print $ofh_counts "$acc"; + print $ofh_TPM "$acc"; + + foreach my $file (@files) { + + my $count = $data{$acc}->{$file}->{count}; + unless (defined $count) { + $count = "NA"; + } + my $tpm = $data{$acc}->{$file}->{TPM}; + if (defined $tpm) { + $tpm = $tpm/1; + } + else { + $tpm = "NA"; + } + + print $ofh_counts "\t$count"; + print $ofh_TPM "\t$tpm"; + } + + print $ofh_counts "\n"; + print $ofh_TPM "\n"; + + } + close $ofh_counts; + close $ofh_TPM; + + ## process gene counts as per txImport-style (Soneson et al. F1000, 2016) + my $gene_counts_file = "$out_prefix.gene.counts.matrix"; + my $gene_tpm_file = "$out_prefix.gene.TPM.not_cross_norm"; + + if ($gene_trans_map_file ne 'none') { + my %gene_to_trans; + { + open(my $fh, $gene_trans_map_file) or die "Error, cannot open file $gene_trans_map_file"; + while (<$fh>) { + chomp; + my ($gene, $trans) = split(/\s+/); + push (@{$gene_to_trans{$gene}}, $trans); + } + close $fh; + } + + open(my $ofh_genecounts, ">$gene_counts_file") or die "Error, cannot write to $gene_counts_file"; + print $ofh_genecounts "\t" . join("\t", @filenames) . "\n"; + + open(my $ofh_genetpm, ">$gene_tpm_file") or die "Error, cannot write to $gene_tpm_file"; + print $ofh_genetpm "\t" . join("\t", @filenames) . "\n"; + + foreach my $gene (sort keys %gene_to_trans) { + my @tpm_vals = ($gene); + my @count_vals = ($gene); + foreach my $file (@filenames) { + my $gene_tpm = 0; + # sum up gene tpm from isoform tpms + foreach my $trans (@{$gene_to_trans{$gene}}) { + my $trans_tpm = $data{$trans}->{ $column_header_to_filename{$file} }->{TPM}; + unless (defined $trans_tpm) { + confess "Error, no TPM value specified for transcript [$trans] of gene [$gene] for sample $file"; + } + $gene_tpm += $trans_tpm; + } + push (@tpm_vals, $gene_tpm); + my $gene_count = $gene_tpm / 1e6 * $sum_sample_counts{ $column_header_to_filename{$file} }; + $gene_count = sprintf("%.2f", $gene_count); + push (@count_vals, $gene_count); + } + + print $ofh_genetpm join("\t", @tpm_vals) . "\n"; + print $ofh_genecounts join("\t", @count_vals) . "\n"; + } + close $ofh_genetpm; + close $ofh_genecounts; + } + if (scalar @files > 1) { + ## more than one sample + + &perform_cross_sample_norm($TPM_matrix_file, "$out_prefix.isoform"); + if ($gene_trans_map_file ne 'none') { + &perform_cross_sample_norm($gene_tpm_file, "$out_prefix.gene"); + } + + } + else { + unless (scalar @files == 1) { + die "Error, no target samples. Shouldn't get here."; + } + print STDERR "Warning, only one sample, so not performing cross-sample normalization\n"; + print STDERR "Done.\n\n"; + } + + exit(0); +} + + +#### +sub perform_cross_sample_norm { + my ($tpm_matrix_file, $out_prefix_name) = @_; + + if ($cross_sample_norm =~ /^TMM$/i) { + my $cmd = "$FindBin::RealBin/support_scripts/run_TMM_scale_matrix.pl --matrix $tpm_matrix_file > $out_prefix_name.$cross_sample_norm.EXPR.matrix"; + &process_cmd($cmd); + } + elsif ($cross_sample_norm =~ /^UpperQuartile$/) { + my $cmd = "$FindBin::RealBin/support_scripts/run_UpperQuartileNormalization_matrix.pl --matrix $tpm_matrix_file > $out_prefix_name.$cross_sample_norm.EXPR.matrix"; + &process_cmd($cmd); + } + elsif ($cross_sample_norm =~ /^none$/i) { + print STDERR "-not performing cross-sample normalization.\n"; + } + +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR $cmd; + my $ret = system($cmd); + if ($ret) { + die "Error, CMD: $cmd died with ret $ret"; + } + + return; +} + + +#### +sub parse_field_positions { + my ($header) = @_; + + my %field_pos; + my @fields = split(/\s+/, $header); + for (my $i = 0; $i <= $#fields; $i++) { + $field_pos{$fields[$i]} = $i; + $field_pos{$i} = $i; # for fixed column assignment + } + + return(%field_pos); +} diff --git a/99.scripts/trinity_utils/util/align_and_estimate_abundance.pl b/99.scripts/trinity_utils/util/align_and_estimate_abundance.pl new file mode 100644 index 0000000..5cdec9a --- /dev/null +++ b/99.scripts/trinity_utils/util/align_and_estimate_abundance.pl @@ -0,0 +1,1015 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; + +use Cwd; +use Carp; + +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use Data::Dumper; + +my %aligner_params = ( + + + ############ + ## Bowtie-1 + ############ + + + 'bowtie_RSEM' => '--all --best --strata -m 300 --chunkmbs 512', + # params used by RSEM itself: + # -a -m 200 + + + 'bowtie_eXpress' => '--all --best --strata -m 300 --chunkmbs 512', + # bowtie -aS -X 800 --offrate 1 (requires: bowtie-build --offrate 1) + + + ############# + ## Bowtie-2 + ############# + + 'bowtie2_RSEM' => '--no-mixed --no-discordant --gbar 1000 --end-to-end -k 200 ', + + ## params used by RSEM itself: + # --dpad 0 --gbar 99999999 --mp 1,1 --np 1 --score-min L,0,-0.1 -I 1 -X 1000 --no-mixed --no-discordant -k 200 + + + 'bowtie2_eXpress' => '--no-mixed --no-discordant --gbar 1000 --end-to-end -k 200 ', + + + # recommended eXpress params: http://bio.math.berkeley.edu/eXpress/faq.html + # -a -X 600 --rdg 6,5 --rfg 6,5 --score-min L,-.6,-.4 --no-discordant --no-mixed + + + 'bowtie_none' => '--all --best --strata -m 300 --chunkmbs 512', + + 'bowtie2_none' => '--no-mixed --no-discordant --gbar 1000 --end-to-end -k 200 ', + + ); + +my $rsem_add_opts = ""; + +my $kallisto_add_opts = ""; +my $salmon_add_opts= ""; + +my $salmon_kmer_length = 31; + +my $usage = <<__EOUSAGE__; + +######################################################################### +# +######################## +# Essential parameters: +######################## +# +# --transcripts transcript fasta file +# +# --seqType fq|fa +# +# If Paired-end: +# +# --left +# --right +# +# or Single-end: +# +# --single +# +# or (preferred): +# +# --samples_file tab-delimited text file indicating biological replicate relationships. +# ex. +# cond_A cond_A_rep1 A_rep1_left.fq A_rep1_right.fq +# cond_A cond_A_rep2 A_rep2_left.fq A_rep2_right.fq +# cond_B cond_B_rep1 B_rep1_left.fq B_rep1_right.fq +# cond_B cond_B_rep2 B_rep2_left.fq B_rep2_right.fq +# +# # if single-end instead of paired-end, then leave the 4th column above empty. +# +# +# +# --est_method abundance estimation method. +# alignment_based: RSEM +# alignment_free: kallisto|salmon +# +################################### +# Potentially optional parameters: +################################### +# +# --output_dir write all files to output directory +# (note, if using --samples_file, output_dir will be set automatically according to replicate name)) +# +# +# if alignment_based est_method: +# --aln_method bowtie|bowtie2 alignment method. (note: RSEM requires either bowtie or bowtie2) +# +########### +# Optional: +# ######### +# +# --SS_lib_type strand-specific library type: paired('RF' or 'FR'), single('F' or 'R'). +# +# --samples_idx restricte processing to sample entry (index starts at one) +# +# +# --thread_count number of threads to use (default = 4) +# +# --debug retain intermediate files +# +# --gene_trans_map file containing 'gene(tab)transcript' identifiers per line. +# or +# --trinity_mode Setting --trinity_mode will automatically generate the gene_trans_map and use it. +# +# +# --prep_reference prep reference (builds target index) +# +# +######################################## +# +# Parameters for single-end reads: +# +# --fragment_length specify RNA-Seq fragment length (default: 200) +# --fragment_std fragment length standard deviation (defalt: 80) +# +######################################## +# +# bowtie-related parameters: (note, tool-specific settings are further below) +# +# --max_ins_size maximum insert size (bowtie -X parameter, default: 800) +# --coordsort_bam provide coord-sorted bam in addition to the default (unsorted) bam. +# +######################################## +# RSEM opts: +# +# --bowtie_RSEM if using 'bowtie', default: \"$aligner_params{bowtie_RSEM}\" +# --bowtie2_RSEM if using 'bowtie2', default: \"$aligner_params{bowtie2_RSEM}\" +# ** if you change the defaults, specify the full set of parameters to use! ** +# +# --include_rsem_bam provide the RSEM enhanced bam file including posterior probabilities of read assignments. +# --rsem_add_opts additional parameters to pass on to rsem-calculate-expression +# +########################################################################## +# kallisto opts: +# +# --kallisto_add_opts default: $kallisto_add_opts +# +########################################################################## +# +# salmon opts: +# +# --salmon_add_opts default: $salmon_add_opts +# +# +# Example usage +# +# ## Just prepare the reference for alignment and abundance estimation +# +# $0 --transcripts Trinity.fasta --est_method salmon --trinity_mode --prep_reference +# +# ## Run the alignment and abundance estimation (assumes reference has already been prepped, errors-out if prepped reference not located.) +# +# $0 --transcripts Trinity.fasta --seqType fq --left reads_1.fq --right reads_2.fq --est_method salmon --trinity_mode --output_dir salmon_quant +# +## ## prep the reference and run the alignment/estimation +# +# $0 --transcripts Trinity.fasta --seqType fq --left reads_1.fq --right reads_2.fq --est_method salmon --trinity_mode --prep_reference --output_dir salmon_quant +# +# ## Use a samples.txt file: +# +# $0 --transcripts Trinity.fasta --est_method salmon --prep_reference --trinity_mode --samples_file samples.txt --seqType fq +# +######################################################################### + + +__EOUSAGE__ + + ; + + + + +my $output_dir; +my $help_flag; +my $transcripts; +my $bam_file; +my $DEBUG_flag = 0; +my $SS_lib_type; +my $thread_count = 4; +my $seqType; +my $left; +my $right; +my $single; +my $gene_trans_map_file; +my $max_ins_size = 800; + +my $est_method; +my $aln_method = ""; + +my $retain_sorted_bam_file = 0; + +my $fragment_length = 200; +my $fragment_std = 80; + +my $output_prefix = ""; + +# devel opts +my $prep_reference = 0; + +my $trinity_mode; + +my $include_rsem_bam; +my $coordsort_bam_flag = 0; + +my $samples_file = ""; +my $samples_idx = 0; + +&GetOptions ( 'help|h' => \$help_flag, + 'transcripts=s' => \$transcripts, + 'name_sorted_bam=s' => \$bam_file, + 'debug' => \$DEBUG_flag, + 'SS_lib_type=s' => \$SS_lib_type, + + 'thread_count=i' => \$thread_count, + + 'gene_trans_map=s' => \$gene_trans_map_file, + 'trinity_mode' => \$trinity_mode, + + 'seqType=s' => \$seqType, + 'left=s' => \$left, + 'right=s' => \$right, + 'single=s' => \$single, + 'max_ins_size=i' => \$max_ins_size, + 'samples_file=s' => \$samples_file, + 'samples_idx=i' => \$samples_idx, + + 'output_dir=s' => \$output_dir, + + 'est_method=s' => \$est_method, + 'aln_method=s' => \$aln_method, + + + 'include_rsem_bam' => \$include_rsem_bam, + + #'output_prefix=s' => \$output_prefix, + + ## devel opts + 'prep_reference' => \$prep_reference, + + # opts for single-end reads + 'fragment_length=i' => \$fragment_length, + 'fragment_std=i' => \$fragment_std, + + # + 'bowtie_RSEM=s' => \($aligner_params{'bowtie_RSEM'}), + 'bowtie2_RSEM=s' => \($aligner_params{'bowtie2_RSEM'}), + + + 'rsem_add_opts=s' => \$rsem_add_opts, + 'kallisto_add_opts=s' => \$kallisto_add_opts, + 'salmon_add_opts=s' => \$salmon_add_opts, + + 'coordsort_bam' => \$coordsort_bam_flag, + + 'salmon_kmer_length=i' => \$salmon_kmer_length, + + ); + + + +if (@ARGV) { + die "Error, don't understand arguments: @ARGV "; +} + +if ($help_flag) { + die $usage; +} + +unless ($est_method) { + die $usage; +} + +my @EST_METHODS = qw(RSEM kallisto salmon); +my %ALIGNMENT_BASED_EST_METHODS = map { + $_ => 1 } qw (RSEM); +my %ALIGNMENT_FREE_EST_METHODS = map { + $_ => 1 } qw (kallisto salmon); + + +unless ( + + ($est_method && $prep_reference && $transcripts && (! ($single||$left||$right||$samples_file)) ) ## just prep reference + + || + + ($transcripts && $est_method && $seqType && ($single || ($left && $right) || $samples_file)) # do alignment + + ) { + + die "Error, missing parameter. See example usage options below.\n" . $usage; +} + + +if ($ALIGNMENT_FREE_EST_METHODS{$est_method}) { + $aln_method = "none"; +} +elsif ($aln_method !~ /bowtie2?/) { + die "Error, --aln_method must be either 'bowtie' or 'bowtie2' "; +} + + +unless ($est_method =~ /^(RSEM|kallisto|salmon|none)$/i) { + die "Error, --est_method @EST_METHODS only\n"; +} + + +my @samples_to_process; +if ($samples_file) { + @samples_to_process = &parse_samples_file($samples_file); + if ($samples_idx > 0) { + my $num_samples = scalar(@samples_to_process); + if ($samples_idx > $num_samples) { + die "Error, sample index $samples_idx > $num_samples num samples "; + } + @samples_to_process = ($samples_to_process[$samples_idx-1]); # run only that sample + } +} +elsif ( ($left && $right) || $single) { + + unless ($output_dir) { + die "Error, must specify output directory name via: --output_dir "; + } + @samples_to_process = &create_sample_definition($output_dir, $left, $right, $single); + +} + + +my $PE_mode = 1; + +if ($single || (@samples_to_process && $samples_to_process[0]->{single})) { + + unless ($fragment_length) { + die "Error, specify --fragment_length for single-end reads (note, not the length of the read but the mean fragment length)\n\n"; + } + + $PE_mode = 0; +} + + +$transcripts = &create_full_path($transcripts); + +$gene_trans_map_file = &create_full_path($gene_trans_map_file) if $gene_trans_map_file; +if ($gene_trans_map_file && ! -s $gene_trans_map_file) { + die "Error, $gene_trans_map_file doesn't exist or is empty"; +} + + +if ($SS_lib_type) { + unless ($SS_lib_type =~ /^(RF|FR|R|F)$/) { + die "Error, do not recognize SS_lib_type: [$SS_lib_type]\n"; + } + if ($PE_mode && length($SS_lib_type) != 2 ) { + die "Error, SS_lib_type [$SS_lib_type] is not compatible with paired reads"; + } +} + +if ( $thread_count !~ /^\d+$/ ) { + die "Error, --thread_count value must be an integer"; +} + + +{ # check for required tools in PATH + + my $missing = 0; + my @tools = ('samtools'); + if ($aln_method eq 'bowtie') { + push (@tools, 'bowtie-build', 'bowtie'); + } + elsif ($aln_method eq 'bowtie2') { + push (@tools, 'bowtie2', 'bowtie2-build'); + } + + if ($est_method =~ /^RSEM$/i) { + push (@tools, 'rsem-calculate-expression'); + } + elsif ($est_method eq 'kallisto') { + push (@tools, 'kallisto'); + } + elsif ($est_method eq 'salmon') { + push (@tools, 'salmon'); + } + + + foreach my $tool (@tools) { + my $p = `sh -c "command -v $tool"`; + unless ($p =~ /\w/) { + warn("ERROR, cannot find $tool in PATH setting: $ENV{PATH}\n\n"); + $missing = 1; + } + } + if ($missing) { + die "Please be sure the utilities @tools are available via your PATH setting.\n"; + } +} + + + +main: { + + if ($trinity_mode && ! $gene_trans_map_file) { + $gene_trans_map_file = "$transcripts.gene_trans_map"; + my $cmd = "$FindBin::RealBin/support_scripts/get_Trinity_gene_to_trans_map.pl $transcripts > $gene_trans_map_file"; + &process_cmd($cmd) unless (-e $gene_trans_map_file); + } + + + + if ($ALIGNMENT_BASED_EST_METHODS{$est_method}) { + + &run_alignment_BASED_estimation(@samples_to_process); + + } + else { + &run_alignment_FREE_estimation(@samples_to_process); + } + + exit(0); +} + + + +#### +sub run_alignment_FREE_estimation { + my @samples = @_; + + + if ($est_method eq "kallisto") { + &run_kallisto(@samples); + } + elsif ($est_method eq "salmon") { + &run_salmon(@samples); + } + else { + die "Error, not recognizing est_method: $est_method"; + # sholdn't get here + } +} + + + +#### +sub run_alignment_BASED_estimation { + my @samples = @_; + + + my $db_index_name = "$transcripts.${aln_method}"; + + + ############################################### + ## Prepare transcript database for alignments + ############################################### + + + if ($prep_reference) { + + my $cmd = "${aln_method}-build $transcripts $db_index_name"; + + unless (-e "$db_index_name.ok") { + + if (-e "$db_index_name.started") { + print STDERR "WARNING - looks like the prep for $db_index_name was already started by another process. Proceeding with caution.\n"; + } + + &process_cmd("touch $db_index_name.started"); + + &process_cmd($cmd); + + rename("$db_index_name.started", "$db_index_name.ok"); + + } + + + } + + if (! -e "$db_index_name.ok") { + die "Error, index $db_index_name not prepared. Be sure to include parameter '--prep_reference' to first prepare the reference for alignment."; + } + + + + my $rsem_prefix = &create_full_path("$transcripts.RSEM"); + + if ($est_method eq 'RSEM') { + + if ($prep_reference) { + + if (-e "$rsem_prefix.rsem.prepped.started") { + print STDERR "WARNING - appears that another process has started the rsem-prep step... proceeding with caution.\n"; + } + + unless (-e "$rsem_prefix.rsem.prepped.ok") { + + &process_cmd("touch $rsem_prefix.rsem.prepped.started"); + + my $cmd = "rsem-prepare-reference "; #--no-bowtie"; # update for RSEM-2.15 + + if ($gene_trans_map_file) { + $cmd .= " --transcript-to-gene-map $gene_trans_map_file"; + } + $cmd .= " $transcripts $rsem_prefix"; + + &process_cmd($cmd); + + rename("$rsem_prefix.rsem.prepped.started", "$rsem_prefix.rsem.prepped.ok"); + } + + + unless (-e "$rsem_prefix.rsem.prepped.ok") { + + die "Error, the RSEM data must first be prepped. Please rerun with '--prep_reference' parameter.\n"; + + } + } + + } + + + unless (@samples) { + print STDERR "Only prepping reference. Stopping now.\n"; + exit(0); + } + + print STDERR Dumper(\@samples); + + my $curr_workdir = cwd(); + foreach my $sample_href (@samples) { + chdir $curr_workdir or die "Error, cannot cd to $curr_workdir"; + # process below will cd into output dir + &run_alignment_do_quant($sample_href, $db_index_name, $rsem_prefix); + } + +} + +#### +sub run_alignment_do_quant { + my ($sample_href, $db_index_name, $rsem_prefix) = @_; + + my $output_dir = $sample_href->{output_dir}; + + ##################### + ## Run alignments + ##################### + + unless (-d $output_dir) { + system("mkdir -p $output_dir"); + } + chdir $output_dir or die "Error, cannot cd to output directory $output_dir"; + + my $prefix = $output_prefix; + if ($prefix) { + $prefix .= "."; # add separator in filename + } + my $bam_file = "${prefix}${aln_method}.bam"; + my $bam_file_ok = "$bam_file.ok"; + + + my $read_type = ($seqType eq "fq") ? "-q" : "-f"; + + ############## + ## Align reads + + my $bowtie_cmd; + + if ($aln_method eq 'bowtie') { + if ($PE_mode) { + my ($left_file, $right_file) = ($sample_href->{left}, $sample_href->{right}); + ## PE alignment + $bowtie_cmd = "set -o pipefail && bowtie $read_type " . $aligner_params{"${aln_method}_${est_method}"} . " -X $max_ins_size -S -p $thread_count $db_index_name -1 $left_file -2 $right_file | samtools view -@ $thread_count -F 4 -S -b | samtools sort -@ $thread_count -n -o $bam_file "; + + } + else { + my $single_file = $sample_href->{single}; + # SE alignment + $bowtie_cmd = "set -o pipefail && bowtie $read_type " . $aligner_params{"${aln_method}_${est_method}"} . " -S -p $thread_count $db_index_name $single_file | samtools view -@ $thread_count -F 4 -S -b | samtools sort -@ $thread_count -n -o $bam_file "; + } + } + elsif ($aln_method eq 'bowtie2') { + + if ($PE_mode) { + ## PE alignment + my ($left_file, $right_file) = ($sample_href->{left}, $sample_href->{right}); + $bowtie_cmd = "set -o pipefail && bowtie2 " . $aligner_params{"${aln_method}_${est_method}"} . " $read_type -X $max_ins_size -x $db_index_name -1 $left_file -2 $right_file -p $thread_count | samtools view -@ $thread_count -F 4 -S -b | samtools sort -@ $thread_count -n -o $bam_file "; + } + else { + # SE alignment + my $single_file = $sample_href->{single}; + $bowtie_cmd = "set -o pipefail && bowtie2 " . $aligner_params{"${aln_method}_${est_method}"} . " $read_type -x $db_index_name -U $single_file -p $thread_count | samtools view -@ $thread_count -F 4 -S -b | samtools sort -@ $thread_count -n -o $bam_file "; + } + } + + &process_cmd($bowtie_cmd) unless (-s $bam_file && -e $bam_file_ok); + + &process_cmd("touch $bam_file_ok") unless (-e $bam_file_ok); + + + if ($est_method eq "RSEM") { + + # convert bam file for use with rsem: + &process_cmd("convert-sam-for-rsem -p $thread_count $bam_file $bam_file.for_rsem"); + + &run_RSEM("$bam_file.for_rsem.bam", $rsem_prefix, $output_prefix); + } + elsif ($est_method eq "none") { + print STDERR "Not running abundance estimation, stopping now after alignment.\n"; + } + else { + die "Error, --est_method $est_method is not supported"; + } + + if ($coordsort_bam_flag) { + + &sort_bam_file($bam_file); + + } + + return; + +} + + +#### +sub sort_bam_file { + my ($bam_file) = @_; + my $sorted_bam_file = $bam_file; + $sorted_bam_file =~ s/bam$/csorted/; + if (! -e "$sorted_bam_file.bam.ok") { + ## sort the bam file + + my $cmd = "samtools sort $bam_file -o $sorted_bam_file.bam"; + &process_cmd($cmd); + $cmd = "samtools index $sorted_bam_file.bam"; + &process_cmd($cmd); + + &process_cmd("touch $sorted_bam_file.bam.ok"); + } + + return; +} + + +#### +sub run_RSEM { + my ($bam_file, $rsem_prefix, $output_prefix) = @_; + + + unless ($output_prefix) { + $output_prefix = "RSEM"; + } + + my $keep_intermediate_files_opt = ($DEBUG_flag) ? "--keep-intermediate-files" : ""; + + my $fraglength_info_txt = ""; + if ($single) { + $fraglength_info_txt = "--fragment-length-mean $fragment_length --fragment-length-sd $fragment_std"; + } + + my $SS_opt = ""; + if ($SS_lib_type) { + if ($SS_lib_type =~ /^F/) { + $SS_opt = "--forward-prob 1.0"; + } + else { + $SS_opt = "--forward-prob 0"; + } + } + + my $no_qualities_string = ""; + if ($seqType eq 'fa') { + $no_qualities_string = "--no-qualities"; + } + + my $paired_flag_text = ($PE_mode) ? "--paired-end" : ""; + + my $rsem_bam_flag = ($include_rsem_bam) ? "" : "--no-bam-output"; + + + my $cmd = "rsem-calculate-expression $no_qualities_string " + . "$paired_flag_text " + . " $rsem_add_opts " + . "-p $thread_count " + . "$fraglength_info_txt " + . "$keep_intermediate_files_opt " + . "$SS_opt $rsem_bam_flag " + . "--bam $bam_file " + . "$rsem_prefix " + . "$output_prefix "; + + unless (-e "$output_prefix.isoforms.results.ok") { + &process_cmd($cmd); + } + &process_cmd("touch $output_prefix.isoforms.results.ok"); + + return; +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + unless ($cmd) { + confess "Error, no cmd specified"; + } + + print STDERR "CMD: $cmd\n"; + + my $ret = system("bash", "-o", "pipefail", "-c", $cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret: $ret"; + } + + return; +} + +### +sub create_full_path { + my ($file_list) = shift; + + my $cwd = cwd(); + + my @files; + + foreach my $file (split(/,/, $file_list)) { + + + if ($file !~ m|^/|) { # must be a relative path + $file = $cwd . "/$file"; + } + + push (@files, $file); + } + + $file_list = join(",", @files); + + return($file_list); + + +} + +#### +sub add_zcat_gz { + my ($file_listing) = @_; + + my @files; + + foreach my $file (split(/,/, $file_listing)) { + + if ($file =~ /\.gz$/) { + + $file = "<(gunzip -c $file)"; # used to be zcat + + + } + push (@files, $file); + } + + $file_listing = join(",", @files); + + return($file_listing); +} + + +#### +sub run_kallisto { + my @samples = @_; + + my $kallisto_index = "$transcripts.kallisto_idx"; + + if ( (! $prep_reference) && (! -e $kallisto_index)) { + confess "Error, no kallisto index file: $kallisto_index, and --prep_reference not set. Re-run with --prep_reference"; + } + if ($prep_reference && ! -e $kallisto_index) { + + my $cmd = "kallisto index -i $kallisto_index $transcripts"; + &process_cmd($cmd); + } + + + if ($SS_lib_type) { + # add strand-specific options for kallisto + my $kallisto_ss_opt = ($SS_lib_type =~ /^R/) ? "--rf-stranded" : "--fr-stranded"; + if ($kallisto_add_opts !~ /$kallisto_ss_opt/) { + $kallisto_add_opts .= " $kallisto_add_opts"; + } + } + + foreach my $sample_href (@samples) { + + my ($output_dir, $left_file, $right_file, $single_file) = ($sample_href->{output_dir}, + $sample_href->{left}, + $sample_href->{right}, + $sample_href->{single}); + + if ($left_file && $right_file) { + + my $cmd = "kallisto quant -i $kallisto_index $kallisto_add_opts -o $output_dir $left_file $right_file"; + &process_cmd($cmd); + } + elsif ($single_file) { + my $cmd = "kallisto quant -l $fragment_length -s $fragment_std -i $kallisto_index -o $output_dir $kallisto_add_opts --single $single_file"; + &process_cmd($cmd); + } + + + if ($gene_trans_map_file) { + + my $cmd = "$FindBin::RealBin/support_scripts/kallisto_trans_to_gene_results.pl $output_dir/abundance.tsv $gene_trans_map_file > $output_dir/abundance.tsv.genes"; + &process_cmd($cmd); + } + } + + return; +} + + + +#### +sub run_salmon { + my (@samples) = @_; + + my $salmon_index = "$transcripts.salmon.idx"; + + if ( (! $prep_reference) && (! -e $salmon_index)) { + confess "Error, no salmon index file: $salmon_index, and --prep_reference not set. Re-run with --prep_reference"; + } + if ($prep_reference && ! -e $salmon_index) { + + ## Prep salmon index + my $cmd = "salmon index -t $transcripts --keepDuplicates -i $salmon_index -k $salmon_kmer_length -p $thread_count"; + + &process_cmd($cmd); + } + + my $num_failures = 0; + + foreach my $sample_href (@samples) { + + my ($output_dir, $left_file, $right_file, $single_file) = ($sample_href->{output_dir}, + $sample_href->{left}, + $sample_href->{right}, + $sample_href->{single}); + + + + my $outdir = $output_dir; #"$output_dir.$salmon_idx_type"; + + if (-s "$outdir/quant.sf") { + print STDERR "-output already exists: $outdir/quant.sf, skipping.\n"; + next; + } + + + eval { + + if ($left_file && $right_file) { + ## PE mode + my $libtype = ($SS_lib_type) ? "IS" . substr($SS_lib_type, 0, 1) : "IU"; + + my $cmd = "salmon quant -i $salmon_index -l $libtype -1 $left_file -2 $right_file -o $outdir $salmon_add_opts -p $thread_count --validateMappings "; + + &process_cmd($cmd); + + } + elsif ($single_file) { + my $libtype = ($SS_lib_type) ? "S" . substr($SS_lib_type, 0, 1) : "U"; + my $cmd = "salmon quant -i $salmon_index -l $libtype -r $single_file -o $outdir $salmon_add_opts -p $thread_count --validateMappings "; + &process_cmd($cmd); + + } + + if ($gene_trans_map_file) { + + my $cmd = "$FindBin::RealBin/support_scripts/salmon_trans_to_gene_results.pl $output_dir/quant.sf $gene_trans_map_file > $output_dir/quant.sf.genes"; + &process_cmd($cmd); + } + }; + if ($@) { + $num_failures++; + print STDERR "Error detected: $@"; + } + } + + + if ($num_failures) { + die "Error, encountered $num_failures failed salmon jobs. See errors above"; + } + + return; +} + + + +#### +sub parse_samples_file { + my ($samples_file) = @_; + + my @samples_to_process; + + my %seen; + open (my $fh, $samples_file) or die "Error, cannot open file: [$samples_file]"; + while (<$fh>) { + chomp; + if (/^\#/) { next; } + unless (/\w/) { next; } + if (/^\-/) { next; } + s/^\s+|\s+$//g; # trim trailing ws + my @x = split(/\s+/); + + my $sample_name = $x[0]; + my $rep_name = $x[1]; + if ($seen{$rep_name}) { + die "Error, replicate names must be unique. Found $rep_name listed multiple times"; + } + $seen{$rep_name}++; + + my $output_dir = $rep_name; + + my $left_fq = $x[2]; + my $right_fq = $x[3]; + + if ($left_fq) { + unless (-s $left_fq) { + die "Error, cannot locate file: $left_fq as specified in samples file: $samples_file"; + } + $left_fq = &create_full_path($left_fq); + if ($left_fq =~ /\.gz$/) { + $left_fq = &add_zcat_gz($left_fq) if ($aln_method eq "bowtie"); + } + } + else { + die "Error, cannot parse line $_ of samples file: $samples_file . See usage info for samples file formatting requirements."; + + } + if ($right_fq) { + unless (-s $right_fq) { + die "Error, cannot locate file $right_fq as specified in samples file: $samples_file"; + } + $right_fq = &create_full_path($right_fq); + if ($right_fq =~ /\.gz$/) { + $right_fq = &add_zcat_gz($right_fq) if ($aln_method eq "bowtie"); + } + } + + if ($left_fq && $right_fq) { + + push (@samples_to_process, { left => $left_fq, + right => $right_fq, + output_dir => $output_dir, + } ); + } + else { + push (@samples_to_process, { single => $left_fq, + output_dir => $output_dir, + } ); + } + + } + + + return (@samples_to_process); +} + + +#### +sub create_sample_definition { + my ($output_dir, $left, $right, $single) = @_; + + $left = &create_full_path($left) if $left; + $right = &create_full_path($right) if $right; + $single = &create_full_path($single) if $single; + + if ($left && $left =~ /\.gz$/) { + $left = &add_zcat_gz($left) if ($aln_method eq "bowtie"); + } + if ($right && $right =~ /\.gz$/) { + $right = &add_zcat_gz($right) if ($aln_method eq "bowtie"); + } + if ($single && $single =~ /\.gz$/) { + $single = &add_zcat_gz($single) if ($aln_method eq "bowtie"); + } + + + if ($left && $right) { + return ( { left => $left, + right => $right, + output_dir => $output_dir, + } ); + } + else { + return( { single => $single, + output_dir => $output_dir, + } ); + } + +} diff --git a/99.scripts/trinity_utils/util/analyze_blastPlus_topHit_coverage.pl b/99.scripts/trinity_utils/util/analyze_blastPlus_topHit_coverage.pl new file mode 100644 index 0000000..ca9daf0 --- /dev/null +++ b/99.scripts/trinity_utils/util/analyze_blastPlus_topHit_coverage.pl @@ -0,0 +1,250 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../PerlLib"); +use Fasta_reader; +use Data::Dumper; + +=ExampleCommands + +# make blastable +makeblastdb \ + -in refTranscripts.fasta\ + -out refTranscripts -dbtype nucl + +# run blast+ +blastn -query Trinity.fasta -db refTranscripts -out blastn.fmt6.txt \ + -evalue 1e-20 -dust no -task megablast -num_threads 2 -max_target_seqs 1 -outfmt 6 + +# analyze results +analyze_blastPlus_topHit_coverage.pl blastn.fmt6.txt refTranscripts.fasta Trinity.fasta + +=cut + + ; + + +my $usage = "usage: $0 blast+.outfmt6.txt query.fasta search_db.fasta [output_prefix=NameOfBlastFileHere] [verbose=0]\n\n"; + +my $blast_out = $ARGV[0] or die $usage; +my $fasta_file_A = $ARGV[1] or die $usage; +my $fasta_file_B = $ARGV[2] or die $usage; # the fasta files don't have to be in any special order. +my $output_prefix = $ARGV[3] || "$blast_out"; +my $verbose = $ARGV[4] || 0; + +main: { + + + my $counter = 0; + + my %query_to_top_hit; # only storing the hit with the greatest blast score. + + # outfmt6: + # qseqid sseqid pident length mismatch gapopen qstart qend sstart send evalue bitscore + + print STDERR "-parsing blast output: $blast_out\n" if $verbose; + open (my $fh, $blast_out) or die "Error, cannot open file $blast_out"; + while (<$fh>) { + chomp; + my $line = $_; + my @x = split(/\t/); + my $query_id = $x[0]; + my $db_id = $x[1]; + my $percent_id = $x[2]; + my $query_start = $x[6]; + my $query_end = $x[7]; + my $db_start = $x[8]; + my $db_end = $x[9]; + + my $evalue = $x[10]; + my $bitscore = $x[11]; + + if ( (! exists $query_to_top_hit{$query_id}) || ($bitscore > $query_to_top_hit{$query_id}->{bitscore}) ) { + + $query_to_top_hit{$query_id} = { query_id => $query_id, + db_id => $db_id, + percent_id => $percent_id, + query_start => $query_start, + query_end => $query_end, + db_start => $db_start, + db_end => $db_end, + evalue => $evalue, + bitscore => $bitscore, + + query_match_len => abs($query_end - $query_start) + 1, + db_match_len => abs($db_end - $db_start) + 1, + + line => $line, + }; + + } + + $counter++; + if ($counter % 100 == 0) { + print STDERR "\r[$counter] " if $verbose; + } + + + } + close $fh; + $counter = 0; + print STDERR "\n" if $verbose; + + ## identify those entries we need sequence length info for. + my %seq_lengths; + my %seq_headers; + { + foreach my $entry (values %query_to_top_hit) { + my $query_id = $entry->{query_id}; + my $db_id = $entry->{db_id}; + + $seq_lengths{$query_id} = undef; + $seq_lengths{$db_id} = undef; + } + + ## get sequence length info + foreach my $fasta_file ($fasta_file_A, $fasta_file_B) { + + print STDERR "-parsing seq length info from file: $fasta_file\n" if $verbose; + + my $fasta_reader = new Fasta_reader($fasta_file); + + while (my $seq_obj = $fasta_reader->next()) { + + $counter++; + if ($counter % 100 == 0) { + print STDERR "\r[$counter] " if $verbose; + } + + + my $acc = $seq_obj->get_accession(); + if (exists $seq_lengths{$acc}) { + + my $sequence = $seq_obj->get_sequence(); + $seq_lengths{$acc} = length($sequence); + + my $header = $seq_obj->get_header(); + # remove the accession + my @header_pieces = split(/\s+/, $header); + shift @header_pieces; + $header = join(" ", @header_pieces); + $seq_headers{$acc} = $header; + } + } + + $counter = 0; + print STDERR "\n" if $verbose; + + } + + } + + + ## analyze the results. + ## make this hit-centric, only retain the longest-coverage query hit for each database sequence. + + print STDERR "-analyzing hits.\n" if $verbose; + my %db_id_to_greatest_pct_cov; # ties broken by bitscore + { + + + open (my $ofh, ">$output_prefix.w_pct_hit_length") or die $!; + + print $ofh join("\t", "#qseqid", "sseqid", "pident", "length", "mismatch", + "gapopen", "qstart", "qend", "sstart", "send", "evalue", "bitscore", + "db_hit_len", "pct_hit_len_aligned", "hit_descr") . "\n"; + + foreach my $entry (values %query_to_top_hit) { + + my $db_id = $entry->{db_id}; + my $db_match_len = $entry->{db_match_len}; + my $db_seq_len = $seq_lengths{$db_id} or die "Error, no length found for $db_id, with hit: " . Dumper($entry); + + my $percent_length_matched = sprintf("%.2f", $db_match_len / $db_seq_len * 100); + + my $line = $entry->{line}; + my $header = $seq_headers{$db_id}; + + print $ofh join("\t", $line, $db_match_len, $percent_length_matched, $header) . "\n"; + + $entry->{db_hit_pct_cov} = $percent_length_matched; + + + if ( ! exists $db_id_to_greatest_pct_cov{$db_id} ) { + $db_id_to_greatest_pct_cov{$db_id} = $entry; + } + else { + my $prev_entry = $db_id_to_greatest_pct_cov{$db_id}; + if ($percent_length_matched > $prev_entry->{db_hit_pct_cov} + || + ($percent_length_matched == $prev_entry->{db_hit_pct_cov} + && + $entry->{bitscore} > $prev_entry->{bitscore}) + ) { + $db_id_to_greatest_pct_cov{$db_id} = $entry; + } + } + + $counter++; + if ($counter % 100 == 0) { + print STDERR "\r[$counter] " if $verbose; + } + + } + close $ofh; + + } + + $counter = 0; + print STDERR "\n" if $verbose; + + ## histogram summary + + my @bins = qw(10 20 30 40 50 60 70 80 90 100); + my %bin_counts; + + open (my $ofh, ">$output_prefix.hist") or die "Error, cannot write to $output_prefix.hist"; + open (my $list_ofh, ">$output_prefix.hist.list") or die $!; + { + + + foreach my $entry (values %db_id_to_greatest_pct_cov) { + + my $pct_cov = $entry->{db_hit_pct_cov}; + + my $prev_bin = 0; + foreach my $bin (@bins) { + if ($pct_cov > $prev_bin && $pct_cov <= $bin) { + $bin_counts{$bin}++; + print $list_ofh join("\t", "Bin_$bin", $entry->{line}) . "\n"; + } + $prev_bin = $bin; + } + + + } + } + close $list_ofh; + + ## Report counts per bin + print "#hit_pct_cov_bin\tcount_in_bin\t>bin_below\n"; + print $ofh "#hit_pct_cov_bin\tcount_in_bin\t>bin_below\n"; + + my $cumul = 0; + foreach my $bin (reverse(@bins)) { + my $count = $bin_counts{$bin} || 0; + $cumul += $count; + print join("\t", $bin, $count, $cumul) . "\n"; + print $ofh join("\t", $bin, $count, $cumul) . "\n"; + + } + close $ofh; + + + exit(0); + + +} diff --git a/99.scripts/trinity_utils/util/filter_low_expr_transcripts.pl b/99.scripts/trinity_utils/util/filter_low_expr_transcripts.pl new file mode 100644 index 0000000..0347117 --- /dev/null +++ b/99.scripts/trinity_utils/util/filter_low_expr_transcripts.pl @@ -0,0 +1,297 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use Getopt::Long qw(:config posix_default no_ignore_case bundling pass_through); +use FindBin; +use lib ("$FindBin::RealBin/../PerlLib"); +use Fasta_reader; + +my $help_flag; + + +my $usage = <<__EOUSAGE__; + +########################################################################################## +# +# --matrix|m expression matrix (TPM or FPKM, *not* raw counts) +# +# --transcripts|t transcripts fasta file (eg. Trinity.fasta) +# +# +# # expression level filter: +# +# --min_expr_any minimum expression level required across any sample (default: 0) +# +# # Isoform-level filtering +# +# --min_pct_dom_iso minimum percent of dominant isoform expression (default: 0) +# or +# --highest_iso_only only retain the most highly expressed isoform per gene (default: off) +# (mutually exclusive with --min_pct_dom_iso param) +# +# # requires gene-to-transcript mappings +# +# --trinity_mode targets are Trinity-assembled transcripts +# or +# --gene_to_trans_map file containing gene-to-transcript mappings +# (format is: gene(tab)transcript ) +# +######################################################################################### + + +__EOUSAGE__ + + ; + + +my $matrix_file; +my $transcripts_file; +my $min_expr_any = 0; +my $min_pct_dom_iso = 0; +my $highest_iso_only_flag = 0; +my $trinity_mode_flag = 0; +my $gene_to_trans_map_file; + + +&GetOptions ( 'help|h' => \$help_flag, + + 'matrix|m=s' => \$matrix_file, + 'transcripts|t=s' => \$transcripts_file, + + 'min_expr_any=f' => \$min_expr_any, + 'min_pct_dom_iso=i' => \$min_pct_dom_iso, + 'highest_iso_only' => \$highest_iso_only_flag, + + 'trinity_mode' => \$trinity_mode_flag, + 'gene_to_trans_map=s' => \$gene_to_trans_map_file, + + + ); + + +if ($help_flag) { + die $usage; +} + + +unless ($matrix_file && $transcripts_file && + ($min_expr_any || $min_pct_dom_iso || $highest_iso_only_flag) ) { + + die $usage; +} + +if ( ($min_pct_dom_iso || $highest_iso_only_flag) && ! ($trinity_mode_flag || $gene_to_trans_map_file) ) { + die "Error, if --min_pct_dom_iso or --highest_iso_only, must also specify either --trinity_mode or --gene_to_trans_map"; +} + +if ($min_pct_dom_iso && $highest_iso_only_flag) { + die "Error, --min_pct_dom_iso and --highest_iso_only are mutually exclusive parameters. "; +} + + +main: { + + my %expr_vals = &parse_expr_matrix($matrix_file); + + if ($min_pct_dom_iso || $highest_iso_only_flag) { + + my %gene_to_iso_map = ($trinity_mode_flag) + ? &parse_Trinity_gene_mapping($transcripts_file) + : &parse_gene_trans_map_file($gene_to_trans_map_file); + + &add_pct_iso_stats(\%expr_vals, \%gene_to_iso_map); + } + + my $total_records = 0; + my $retained_records = 0; + + my $fasta_reader = new Fasta_reader($transcripts_file); + while (my $seq_obj = $fasta_reader->next()) { + + $total_records++; + + my $acc = $seq_obj->get_accession(); + + my $keep_flag = 1; + + my $info_struct = $expr_vals{$acc} or die "Error, no expression record stored for acc: [$acc]. Be sure to provide the transcript expression matrix and all transcripts in the $transcripts_file must have records in the transcript expression matrix file."; + + if ($min_expr_any && $info_struct->{max_expr} < $min_expr_any) { + $keep_flag = 0; + print STDERR "-excluding $acc, max_expr: $info_struct->{max_expr} < $min_expr_any\n"; + } + if ($min_pct_dom_iso && (! $info_struct->{top_iso}) && $info_struct->{pct_dom_iso_expr} < $min_pct_dom_iso) { + # notice we'll still keep the dominant isoform for the gene even if it's pct iso < $min_pct_dom_iso. + ## dont want to be silly and throw out the gene altogther... :) + print STDERR "-excluding $acc, pct_dom_iso_expr $info_struct->{pct_dom_iso_expr} < $min_pct_dom_iso\n"; + + $keep_flag = 0; + } + + if ($highest_iso_only_flag && ! $info_struct->{top_iso}) { + print STDERR "-excluding $acc, not top_iso\n"; + $keep_flag = 0; + } + + if ($keep_flag) { + $retained_records++; + my $fasta_record = $seq_obj->get_FASTA_format(); + chomp $fasta_record; + my ($header_line, @seq_lines) = split(/\n/, $fasta_record); + # tack on the pct expr info onto the header + my $top_iso_flag = $info_struct->{top_iso}; + my $pct_iso_expr = (defined $info_struct->{pct_iso_expr}) ? $info_struct->{pct_iso_expr} : "NA"; + + my $pct_dom_iso_expr = (defined $info_struct->{pct_dom_iso_expr}) ? $info_struct->{pct_dom_iso_expr} : "NA"; + + $header_line .= " top_iso:$top_iso_flag pct_iso_expr=$pct_iso_expr pct_dom_iso_expr=$pct_dom_iso_expr max_expr_any=$info_struct->{max_expr}"; + + print join("\n", $header_line, @seq_lines) . "\n"; + + } + } + + my $pct_records_retained = sprintf("%.2f", $retained_records / $total_records * 100); + print STDERR "\n\n\tRetained $retained_records / $total_records = $pct_records_retained\% of total transcripts.\n\n\n"; + + + exit(0); + + +} + +#### +sub add_pct_iso_stats { + my ($expr_vals_href, $gene_to_iso_map_href) = @_; + + foreach my $gene (keys %$gene_to_iso_map_href) { + + my @isoforms = keys %{$gene_to_iso_map_href->{$gene}}; + + if (scalar @isoforms == 1) { + # only one isoform, so must be 100% of that gene. + $expr_vals_href->{ $isoforms[0] }->{pct_iso_expr} = 100; + $expr_vals_href->{ $isoforms[0] }->{pct_dom_iso_expr} = 100; + $expr_vals_href->{ $isoforms[0] }->{top_iso} = 1; + + } + else { + # determine fraction of total gene expr + # first, get sum of gene expr across isoforms + my $gene_sum_expr = 0; + my $dominant_iso_expr = 0; + foreach my $iso (@isoforms) { + + my $expr = $expr_vals_href->{$iso}->{sum_expr}; + if (!defined($expr)) { + use Data::Dumper; + print STDERR "ISO: $iso\t" . Dumper($expr_vals_href->{$iso}); + } + if ($expr > $dominant_iso_expr) { + $dominant_iso_expr = $expr; + } + + + $gene_sum_expr += $expr; + } + # now compute pct iso + foreach my $iso (@isoforms) { + my $expr = $expr_vals_href->{$iso}->{sum_expr}; + my $pct_iso = 0; + if ($gene_sum_expr > 0) { + $pct_iso = sprintf("%.2f", $expr / $gene_sum_expr * 100); + } + $expr_vals_href->{$iso}->{pct_iso_expr} = $pct_iso; + + my $pct_dom_iso_expr = 0; + if ($dominant_iso_expr > 0) { + $pct_dom_iso_expr = sprintf("%.2f", $expr / $dominant_iso_expr * 100); + } + $expr_vals_href->{$iso}->{pct_dom_iso_expr} = $pct_dom_iso_expr; + } + # set top iso + @isoforms = sort { $expr_vals_href->{$a}->{pct_iso_expr} <=> $expr_vals_href->{$b}->{pct_iso_expr} } @isoforms; + + my $top_isoform = pop @isoforms; # note, if there's no gene expression for some reason, choice isn't informative. + $expr_vals_href->{$top_isoform}->{top_iso} = 1; + } + } +} + + +#### +sub parse_expr_matrix { + my ($matrix_file) = @_; + + my %expr_vals; + + open (my $fh, $matrix_file) or die "Error, cannot open file $matrix_file"; + my $header = <$fh>; + while (<$fh>) { + chomp; + my @expr = split(/\t/); + my $acc = shift @expr; + + my $max_val = 0; + my $sum = 0; + + foreach my $expr_val (@expr) { + $sum += $expr_val; + if ($expr_val > $max_val) { + $max_val = $expr_val; + } + } + + $expr_vals{$acc}->{max_expr} = $max_val; + $expr_vals{$acc}->{sum_expr} = $sum; + $expr_vals{$acc}->{pct_iso_expr} = undef; # set later + $expr_vals{$acc}->{pct_dom_iso_expr} = undef; + $expr_vals{$acc}->{top_iso} = 0; # set later to the isoform with highest expression for that gene. + + } + close $fh; + + return(%expr_vals); +} + +#### +sub parse_Trinity_gene_mapping { + my ($transcripts_file) = @_; + + my %gene_to_iso_map; + + open (my $fh, $transcripts_file) or die "Error, cannot open file $transcripts_file"; + while (<$fh>) { + if (/^>(\S+)/) { + my $acc = $1; + $acc =~ /^(\S+)(_i\d+)$/ or die "Error, cannot parse Trinity accession: $acc"; + my $gene_id = $1; + + $gene_to_iso_map{$gene_id}->{$acc} = 1; + } + } + close $fh; + + return(%gene_to_iso_map); +} + +#### +sub parse_gene_trans_map_file { + my ($gene_to_trans_map_file) = @_; + + my %gene_to_iso_map; + + open (my $fh, $gene_to_trans_map_file) or die "Error, cannot open file $gene_to_trans_map_file"; + while (<$fh>) { + chomp; + my ($gene, $trans) = split(/\t/); + + $gene_to_iso_map{$gene}->{$trans} = 1; + } + close $fh; + + return (%gene_to_iso_map); +} + diff --git a/99.scripts/trinity_utils/util/insilico_read_normalization.pl b/99.scripts/trinity_utils/util/insilico_read_normalization.pl new file mode 100644 index 0000000..79800d3 --- /dev/null +++ b/99.scripts/trinity_utils/util/insilico_read_normalization.pl @@ -0,0 +1,1050 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use threads; +no strict qw(subs refs); + +use FindBin; +use lib ("$FindBin::RealBin/../PerlLib"); +use File::Basename; +use Cwd; +use Carp; +use Getopt::Long qw(:config no_ignore_case pass_through); +use Fastq_reader; +use Fasta_reader; +use threads; +use Data::Dumper; +use COMMON; +use DB_File; + +$ENV{PATH} = "$FindBin::Bin/../trinity-plugins/BIN:$ENV{PATH}"; + + +open (STDERR, ">&STDOUT"); ## capturing stderr and stdout in a single stdout stream + + +my $SYMLINK = ($ENV{NO_SYMLINK}) ? "cp" : "ln -sf"; + +## Jellyfish +my $max_memory; + +# Note: For the Trinity logo below the backslashes are quoted in order to keep +# them from quoting the character than follows them. "\\" keeps "\ " from occuring. + +my $output_directory = cwd(); +my $help_flag; +my $seqType; +my @left_files; +my @right_files; +my $left_list_file; +my $right_list_file; +my @single_files; +my $SS_lib_type; +my $CPU = 2; +my $MIN_KMER_COV_CONST = 2; ## DO NOT CHANGE +my $max_cov; +my $pairs_together_flag = 0; +my $max_CV = 10000; # effectively turning this off +my $KMER_SIZE = 25; +my $MIN_COV = 0; + +my $__devel_report_kmer_cov_stats = 0; + +my $PARALLEL_STATS = 0; +my $JELLY_S; + +my $NO_SEQTK = 0; + +my $usage = <<_EOUSAGE_; + + +############################################################################### +# +# Required: +# +# --seqType :type of reads: ( 'fq' or 'fa') +# --JM :(Jellyfish Memory) number of GB of system memory to use for +# k-mer counting by jellyfish (eg. 10G) *include the 'G' char +# +# +# --max_cov :targeted maximum coverage for reads. +# +# +# If paired reads: +# --left :left reads (if specifying multiple files, list them as comma-delimited. eg. leftA.fq,leftB.fq,...) +# --right :right reads +# +# Or, if unpaired reads: +# --single :single reads +# +# Or, if you have read collections in different files you can use 'list' files, where each line in a list +# file is the full path to an input file. This saves you the time of combining them just so you can pass +# a single file for each direction. +# --left_list :left reads, one file path per line +# --right_list :right reads, one file path per line +# +#################################### +## Misc: ######################### +# +# --pairs_together :process paired reads by averaging stats between pairs and retaining linking info. +# +# --SS_lib_type :Strand-specific RNA-Seq read orientation. +# if paired: RF or FR, +# if single: F or R. (dUTP method = RF) +# See web documentation. +# --output :name of directory for output (will be +# created if it doesn't already exist) +# default( "${output_directory}" ) +# +# --CPU :number of threads to use (default: = $CPU) +# --PARALLEL_STATS :generate read stats in parallel for paired reads +# +# --KMER_SIZE :default $KMER_SIZE +# +# --max_CV :maximum coeff of var (default: $max_CV) +# +# --min_cov :minimum kmer coverage for a read to be retained (default: $MIN_COV) +# +# --no_cleanup :leave intermediate files +# --tmp_dir_name default("tmp_normalized_reads"); +# +############################################################################### + + + + +_EOUSAGE_ + + ; + +my $ROOTDIR = "$FindBin::RealBin/../"; +my $UTILDIR = "$ROOTDIR/util/support_scripts/"; +my $INCHWORM_DIR = "$ROOTDIR/Inchworm"; + +unless (@ARGV) { + die "$usage\n"; +} + +my $NO_CLEANUP = 0; + +my $TMP_DIR_NAME = "tmp_normalized_reads"; + + +&GetOptions( + + 'h|help' => \$help_flag, + + ## general opts + "seqType=s" => \$seqType, + "left=s{,}" => \@left_files, + "right=s{,}" => \@right_files, + "single=s{,}" => \@single_files, + + "left_list=s" => \$left_list_file, + "right_list=s" => \$right_list_file, + + "SS_lib_type=s" => \$SS_lib_type, + "max_cov=i" => \$max_cov, + "min_cov=i" => \$MIN_COV, + + "output=s" => \$output_directory, + + # Jellyfish + 'JM=s' => \$max_memory, # in GB + + # misc + 'KMER_SIZE=i' => \$KMER_SIZE, + 'CPU=i' => \$CPU, + 'PARALLEL_STATS' => \$PARALLEL_STATS, + 'kmer_size=i' => \$KMER_SIZE, + 'max_CV=i' => \$max_CV, + 'pairs_together' => \$pairs_together_flag, + + 'no_cleanup' => \$NO_CLEANUP, + + #devel + '__devel_report_kmer_cov_stats' => \$__devel_report_kmer_cov_stats, + 'jelly_s=i' => \$JELLY_S, + + 'tmp_dir_name=s' => \$TMP_DIR_NAME, + + "NO_SEQTK" => \$NO_SEQTK, + +); + + + +if ($help_flag) { + die "$usage\n"; +} + +if (@ARGV) { + die "Error, do not understand options: @ARGV\n"; +} + + +unless ($seqType =~ /^(fq|fa)$/) { + die "Error, set --seqType to 'fq' or 'fa'"; +} +unless ($max_memory && $max_memory =~ /^\d+G/) { + die "Error, must set --JM to number of G of RAM (ie. 10G) "; +} + +if ($SS_lib_type) { + unless ($SS_lib_type =~ /^(R|F|RF|FR)$/) { + die "Error, unrecognized SS_lib_type value of $SS_lib_type. Should be: F, R, RF, or FR\n"; + } + + ## note, if single-end reads and strand-specific, just treat it as F and don't bother revcomplementing here (waste of time) + if ($SS_lib_type eq 'R') { + $SS_lib_type = 'F'; + } + +} + + +if ($left_list_file) { + @left_files = &read_list_file($left_list_file); +} +if ($right_list_file) { + @right_files = &read_list_file($right_list_file); +} + +unless (@single_files || (@left_files && @right_files)) { + die "Error, need either options 'left' and 'right' or option 'single'\n"; +} + + +if (@left_files) { + @left_files = split(",", join(",", @left_files)); +} +if (@right_files) { + @right_files = split(",", join(",", @right_files)); +} +if (@single_files) { + @single_files = split(",", join(",", @single_files)); +} + + + +unless ($max_cov && $max_cov >= 2) { + die "Error, need to set --max_cov at least 2"; +} + + + +## keep the original 'xG' format string for the --JM option, then calculate the numerical value for max_memory +my $JM_string = $max_memory; ## this one is used in the Chrysalis exec string +my $sort_mem; +if ($max_memory) { + $max_memory =~ /^([\d\.]+)G$/ or die "Error, cannot parse max_memory value of $max_memory. Set it to 'xG' where x is a numerical value\n"; + + $max_memory = $1; + + # prep the sort memory usage + $sort_mem = $max_memory; + if ($PARALLEL_STATS) { + $sort_mem = int($sort_mem/2); + unless ($sort_mem > 1) { + $sort_mem = 1; + } + } + $sort_mem .= "G"; + + $max_memory *= 1024**3; # convert to from gig to bytes +} +else { + die "Error, must specify max memory for jellyfish to use, eg. --JM 10G \n"; +} + +if ($pairs_together_flag && ! ( @left_files && @right_files) ) { + die "Error, if setting --pairs_together, must use the --left and --right parameters."; +} + + +my $sort_exec = &COMMON::get_sort_exec($CPU); + + + +main: { + + my $start_dir = cwd(); + + ## create complete paths for input files: + @left_files = &create_full_path(@left_files) if @left_files; + @right_files = &create_full_path(@right_files) if @right_files; + $left_list_file = &create_full_path($left_list_file) if $left_list_file; + $right_list_file = &create_full_path($right_list_file) if $right_list_file; + @single_files = &create_full_path(@single_files) if @single_files; + $output_directory = &create_full_path($output_directory); + + unless (-d $output_directory) { + + mkdir $output_directory or die "Error, cannot mkdir $output_directory"; + } + + chdir ($output_directory) or die "Error, cannot cd to $output_directory"; + + my $tmp_directory = "$output_directory/$TMP_DIR_NAME"; + + my $CREATED_TMP_DIR_HERE_FLAG = 0; + if (! -d $tmp_directory) { + mkdir $tmp_directory or die "Error, cannot mkdir $tmp_directory"; + $CREATED_TMP_DIR_HERE_FLAG = 1; + } + chdir $tmp_directory or die "Error, cannot cd to $tmp_directory"; + + my $trinity_target_fa = (@single_files) ? "single.fa" : "both.fa"; + + my @files_need_stats; + my @checkpoints; + + + print STDERR "-prepping seqs\n"; + if ( (@left_files && @right_files) || + ($left_list_file && $right_list_file) ) { + + my ($left_SS_type, $right_SS_type); + if ($SS_lib_type) { + ($left_SS_type, $right_SS_type) = split(//, $SS_lib_type); + } + + print STDERR "Converting input files. (both directions in parallel);"; + + my $thr1; + my $thr2; + + if (-s "left.fa" && -e "left.fa.ok") { + $thr1 = threads->create(sub { print STDERR (" Left file exists, nothing to do;");}); + } + else { + $thr1 = threads->create('prep_list_of_seqs', \@left_files, $seqType, "left", $left_SS_type); + push (@checkpoints, ["left.fa", "left.fa.ok"]); + } + + if (-s "right.fa" && -e "right.fa.ok") { + $thr2 = threads->create(sub { print STDERR (" Right file exists, nothing to do;");}); + } + else { + $thr2 = threads->create('prep_list_of_seqs', \@right_files, $seqType, "right", $right_SS_type); + push (@checkpoints, ["right.fa", "right.fa.ok"]); + } + + $thr1->join(); + $thr2->join(); + + if ($thr1->error() || $thr2->error()) { + die "Error, conversion thread failed"; + } + + &process_checkpoints(@checkpoints); + + print STDERR " Done converting input files. "; + + push (@files_need_stats, + [\@left_files, "left.fa"], + [\@right_files, "right.fa"]); + + @checkpoints = (); + &process_cmd("cat left.fa right.fa > $trinity_target_fa") unless (-s $trinity_target_fa && -e "$trinity_target_fa.ok"); + unless (-s $trinity_target_fa == ((-s "left.fa") + (-s "right.fa"))){ + die "$trinity_target_fa (".(-s $trinity_target_fa)." bytes) is different from the combined size of left.fa and right.fa (".((-s "left.fa") + (-s "right.fa"))." bytes)\n"; + } + push (@checkpoints, [ $trinity_target_fa, "$trinity_target_fa.ok" ]); + + } + elsif (@single_files) { + + ## Single-mode + + unless (-s "single.fa" && -e "single.fa.ok") { + &prep_list_of_seqs(\@single_files, $seqType, "single", $SS_lib_type); + + } + push (@files_need_stats, [\@single_files, "single.fa"]); + push (@checkpoints, [ "single.fa", "single.fa.ok" ]); + + } + else { + die "not sure what to do. "; # should never get here. + } + + &process_checkpoints(@checkpoints); + + print STDERR "-kmer counting.\n"; + my $kmer_file = &run_jellyfish($trinity_target_fa, $SS_lib_type); + + print STDERR "-generating stats files\n"; + &generate_stats_files(\@files_need_stats, $kmer_file, $SS_lib_type); + + print STDERR "-defining normalized reads\n"; + if ($pairs_together_flag) { + &run_nkbc_pairs_together(\@files_need_stats, $kmer_file, $SS_lib_type, $max_cov, $MIN_COV, $max_CV); + } else { + &run_nkbc_pairs_separate(\@files_need_stats, $kmer_file, $SS_lib_type, $max_cov, $MIN_COV, $max_CV); + } + + print STDERR "-search and capture.\n"; + my @outputs; + @checkpoints = (); + my @threads; + my $thread_counter = 0; + foreach my $info_aref (@files_need_stats) { + my ($orig_file, $converted_file, $stats_file, $selected_entries) = @$info_aref; + + ## do multi-threading + + my $base = (scalar @$orig_file == 1) ? basename($orig_file->[0]) : basename($orig_file->[0]) . "_ext_all_reads"; + + my $normalized_filename_prefix = $output_directory . "/$base.normalized_K${KMER_SIZE}_maxC${max_cov}_minC${MIN_COV}_maxCV${max_CV}"; + my $outfile; + + if ($seqType eq 'fq') { + $outfile = "$normalized_filename_prefix.fq"; + } + else { + # fastA + $outfile = "$normalized_filename_prefix.fa"; + } + push (@outputs, $outfile); + + ## run in parallel + + $thread_counter += 1; + my $checkpoint_file = "$outfile.ok"; + unless (-e $checkpoint_file) { + + my $thread = threads->create('make_normalized_reads_file', $orig_file, $seqType, $selected_entries, $outfile, $thread_counter); + + push (@threads, $thread); + push (@checkpoints, [$outfile, $checkpoint_file]); + } + } + + my $num_fail = 0; + foreach my $thread (@threads) { + $thread->join(); + if ($thread->error()) { + print STDERR " Error encountered with thread.\n"; + $num_fail++; + } + } + if ($num_fail) { + die "Error, at least one thread died"; + } + + &process_checkpoints(@checkpoints); + + chdir $output_directory or die "Error, cannot chdir to $output_directory"; + + ## link them up with simpler names so they're easy to find by downstream scripts. + if (scalar @outputs == 2) { + # paired + my $left_out = $outputs[0]; + my $right_out = $outputs[1]; + + &process_cmd("$SYMLINK $left_out left.norm.$seqType"); + &process_cmd("$SYMLINK $right_out right.norm.$seqType"); + } + else { + my $single_out = $outputs[0]; + &process_cmd("$SYMLINK $single_out single.norm.$seqType"); + } + + unless ($NO_CLEANUP) { + if ($CREATED_TMP_DIR_HERE_FLAG) { + print STDERR "-removing tmp dir $tmp_directory\n"; + `rm -rf $tmp_directory`; + } + } + + print STDERR "\n\nNormalization complete. See outputs: \n\t" . join("\n\t", @outputs) . "\n"; + + + exit(0); +} + + +#### +sub build_selected_index { + my ($file, $thread_count) = @_; + + + + my $index_href = {}; + + my $tied_idx_filename = $file . ".thread-${thread_count}.idx"; + if (-s $tied_idx_filename) { + unlink($tied_idx_filename); + } + + tie (%{$index_href}, 'DB_File', $tied_idx_filename, O_CREAT|O_RDWR, 0666, $DB_BTREE); + + + open(my $ifh, $file) || die "failed to read selected_entries file $file: $!"; + + while (my $line = <$ifh> ) { + chomp $line; + next unless $line =~ /\S/; + + ## want core, .... just in case. + $line =~ s|/\w$||; + + #print STDERR "-want $line\n"; + + + $index_href->{$line} = 0; + } + + return ($index_href); +} + + +#### +sub make_normalized_reads_file { + my ($source_files_aref, $seq_type, $selected_entries, $outfile, $thread_count) = @_; + + print STDERR "-preparing to extract selected reads from: @$source_files_aref ..."; + open (my $ofh, ">$outfile") or die "Error, cannot write to $outfile"; + + my @source_files = @$source_files_aref; + + my $idx_href = &build_selected_index( $selected_entries, $thread_count ); + print STDERR " done prepping, now search and capture.\n"; + + #print STDERR Dumper(\%idx); + + for my $orig_file ( @source_files ) { + my $reader; + + print STDERR "-capturing normalized reads from: $orig_file\n"; + + # if we had a consistent interface for the readers, we wouldn't have to code this up separately below... oh well. + ## ^^ I enjoyed this lamentation, so I left it in the rewrite - JO + if ($seqType eq 'fq') { $reader = new Fastq_reader($orig_file) } + elsif ($seqType eq 'fa') { $reader = new Fasta_reader($orig_file) } + else { die "Error, do not recognize format: $seqType" } + + while ( my $seq_obj = $reader->next() ) { + + my $acc; + + if ($seqType eq 'fq') { + $acc = $seq_obj->get_core_read_name(); + $acc =~ s/_forward//; + $acc =~ s/_reverse//; + } + elsif ($seqType eq 'fa') { + $acc = $seq_obj->get_accession(); + $acc =~ s|/[12]\s*$||; + } + + #print STDERR "parsed acc: [$acc]\n"; + + if ( exists $idx_href->{$acc} ) { + $idx_href->{$acc}++; + my $record = ''; + + if ($seqType eq 'fq') { + $record = $seq_obj->get_fastq_record(); + } + elsif ($seqType eq 'fa') { + $record = $seq_obj->get_FASTA_format(fasta_line_len => -1); + } + + print $ofh $record; + } + } + } + + ## check and make sure they were all found + my %missing; + for my $k ( keys %{$idx_href} ) { + if ($idx_href->{$k} == 0) { + + $missing{$k} = 1; + } + + } + close $ofh; + + my $not_found_count = scalar(keys %missing); + + if ( $not_found_count ) { + open (my $ofh, ">$outfile.missing_accs") or die "Error, cannot write to file $outfile.missing_accs"; + print $ofh join("\n", keys %missing) . "\n"; + close $ofh; + + die "Error, not all specified records have been retrieved (missing $not_found_count) from @source_files, see file: $outfile.missing_accs for list of missing entries"; + + } + + return; +} + + +#### +sub run_jellyfish { + my ($reads, $strand_specific_flag) = @_; + + my $jelly_kmer_fa_file = "jellyfish.K${KMER_SIZE}.min${MIN_KMER_COV_CONST}.kmers.fa"; + + my $jellyfish_checkpoint = "$jelly_kmer_fa_file.success"; + + unless (-e $jellyfish_checkpoint) { + + print STDERR "-------------------------------------------\n" + . "----------- Jellyfish --------------------\n" + . "-- (building a k-mer catalog from reads) --\n" + . "-------------------------------------------\n\n"; + + + my $read_file_size = -s $reads; + + my $jelly_hash_size = int( ($max_memory - $read_file_size)/7); # decided upon by Rick Westerman + + + if ($jelly_hash_size < 100e6 || $read_file_size < 5e9) { + $jelly_hash_size = 100e6; # seems reasonable for a min hash size as 100M + } + + ## for testing + if ($JELLY_S) { + $jelly_hash_size = $JELLY_S; + } + + my $cmd = "jellyfish count -t $CPU -m $KMER_SIZE -s $jelly_hash_size "; + + unless ($SS_lib_type) { + ## count both strands + $cmd .= " --canonical "; + } + + $cmd .= " $reads"; + + &process_cmd($cmd); + + + if (-s $jelly_kmer_fa_file) { + unlink($jelly_kmer_fa_file) or die "Error, cannot unlink $jelly_kmer_fa_file"; + } + + my $jelly_db = "mer_counts.jf"; + + ## write a histogram of the kmer counts. + $cmd = "jellyfish histo -t $CPU -o $jelly_kmer_fa_file.histo $jelly_db"; + &process_cmd($cmd); + + + + $cmd = "jellyfish dump -L $MIN_KMER_COV_CONST $jelly_db > $jelly_kmer_fa_file"; + + &process_cmd($cmd); + + unlink($jelly_db); + + ## if got this far, consider jellyfish done. + &process_cmd("touch $jellyfish_checkpoint"); + + } + + + return($jelly_kmer_fa_file); +} + + +#### (from Trinity.pl) +## WARNING: this function appends to the target output file, so a -s check is advised +# before you call this for the first time within any given script. +sub prep_seqs { + my ($initial_file, $seqType, $file_prefix, $SS_lib_type) = @_; + + my $read_type = ($file_prefix eq "right") ? "2" : "1"; + + + my $using_FIFO_flag = 0; + if ($initial_file =~ /\.gz$|\.xz$|\.bz2$/) { + ($initial_file) = &add_fifo_for_gzip($initial_file); + $using_FIFO_flag = 1; + } + + if ($seqType eq "fq") { + # make fasta + + if ($NO_SEQTK) { + my $perlcmd = "$UTILDIR/fastQ_to_fastA.pl -I $initial_file "; + + if ($SS_lib_type && $SS_lib_type eq "R") { + $perlcmd .= " --rev "; + + } + $perlcmd .= " >> $file_prefix.fa 2> $file_prefix.readcount "; + + &process_cmd($perlcmd); + } + else { + # using seqtk + my $cmd = "seqtk-trinity seq -A -R $read_type "; + if ($SS_lib_type && $SS_lib_type eq "R") { + $cmd =~ s/trinity seq /trinity seq -r /; + } + $cmd .= " $initial_file >> $file_prefix.fa"; + + &process_cmd($cmd); + } + } + elsif ($seqType eq "fa") { + if ($SS_lib_type && $SS_lib_type eq "R") { + my $cmd = "$UTILDIR/revcomp_fasta.pl $initial_file >> $file_prefix.fa"; + &process_cmd($cmd); + } + else { + + if ($using_FIFO_flag) { + # can't symlink it, so just cat it + my $cmd = "cat $initial_file >> $file_prefix.fa"; + &process_cmd($cmd); + } + else { + ## just symlink it here: + my $cmd = "cat $initial_file >> $file_prefix.fa"; + &process_cmd($cmd); + } + } + } + + + return; +} + + + +### +sub prep_list_of_seqs { + my ($files, $seqType, $file_prefix, $SS_lib_type) = @_; + + # generates $file_prefix.fa by converting & concatenating $files of $seqType + + eval { + + for my $file ( @$files ) { + prep_seqs( $file, $seqType, $file_prefix, $SS_lib_type); + } + }; + + if ($@) { + # remove faulty output file + unlink("$file_prefix.fa"); + die $@; + } + + return 0; +} + + +### +sub create_full_path { + my (@files) = @_; + + my @ret; + foreach my $file (@files) { + my $cwd = cwd(); + if ($file !~ m|^/|) { # must be a relative path + $file = $cwd . "/$file"; + } + push (@ret, $file); + } + + if (wantarray) { + return(@ret); + } + else { + if (scalar @ret > 1) { + confess("Error, provided multiple files as input, but only requesting one file in return"); + } + return($ret[0]); + } +} + + +### +sub read_list_file { + my ($file, $regex) = @_; + + my @files; + + open(my $ifh, $file) || die "failed to read input list file ($file): $!"; + + while (my $line = <$ifh>) { + chomp $line; + next unless $line =~ /\S/; + + if ( defined $regex ) { + if ( $line =~ /$regex/ ) { + push @files, $line; + } + } else { + push @files, $line; + } + } + + return @files; +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $start_time = time(); + my $ret = system("bash", "-c", $cmd); + my $end_time = time(); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + print STDERR "CMD finished (" . ($end_time - $start_time) . " seconds)\n"; + + return; +} + +#### +sub generate_stats_files { + my ($files_need_stats_aref, $kmer_file, $SS_lib_type) = @_; + + my @cmds; + + + my $CPU_ADJ = $CPU; + if ($PARALLEL_STATS) { + $CPU_ADJ = int($CPU/2); + if ($CPU_ADJ < 1) { + $CPU_ADJ = 1; + } + } + + my @checkpoints; + foreach my $info_aref (@$files_need_stats_aref) { + my ($orig_file, $converted_fa_file) = @$info_aref; + + my $stats_filename = "$converted_fa_file.K$KMER_SIZE.stats"; + push (@$info_aref, $stats_filename); + + my $cmd = "$INCHWORM_DIR/bin/fastaToKmerCoverageStats --reads $converted_fa_file --kmers $kmer_file --kmer_size $KMER_SIZE --num_threads $CPU_ADJ "; + unless ($SS_lib_type) { + $cmd .= " --DS "; + } + + if ($__devel_report_kmer_cov_stats) { + $cmd .= " --capture_coverage_info "; + } + + $cmd .= " > $stats_filename"; + + push (@cmds, $cmd) unless (-e "$stats_filename.ok"); + push (@checkpoints, ["$stats_filename", "$stats_filename.ok"]); + } + + if (@cmds) { + if ($PARALLEL_STATS) { + &process_cmds_parallel(@cmds); + } + else { + &process_cmds_serial(@cmds); + } + } + + &process_checkpoints(@checkpoints); + + + + { + ## sort by read name + my @cmds; + @checkpoints = (); + foreach my $info_aref (@$files_need_stats_aref) { + my $stats_file = $info_aref->[-1]; + my $sorted_stats_file = $stats_file . ".sort"; + + # retain column headers, sort the rest. + my $cmd = "head -n1 $stats_file > $sorted_stats_file && tail -n +2 $stats_file | $sort_exec -k1,1 -T . -S $sort_mem >> $sorted_stats_file"; + push (@cmds, $cmd) unless (-e "$sorted_stats_file.ok"); + $info_aref->[-1] = $sorted_stats_file; + push (@checkpoints, [$sorted_stats_file, "$sorted_stats_file.ok"]); + + } + + if (@cmds) { + print STDERR "-sorting each stats file by read name.\n"; + + if ($PARALLEL_STATS) { + &process_cmds_parallel(@cmds); + } + else { + &process_cmds_serial(@cmds); + } + + &process_checkpoints(@checkpoints); + } + + + } + + return; +} + + +#### +sub process_checkpoints { + my @checkpoints = @_; + + foreach my $checkpoint (@checkpoints) { + my ($outfile, $checkpoint_file) = @$checkpoint; + if (-s "$outfile" && ! -e $checkpoint_file) { + &process_cmd("touch $checkpoint_file"); + } + } + + return; +} + + +#### +sub run_nkbc_pairs_separate { + my ($files_need_stats_aref, $kmer_file, $SS_lib_type, $max_cov, $min_cov, $max_CV) = @_; + + my @cmds; + + my @checkpoints; + foreach my $info_aref (@$files_need_stats_aref) { + my ($orig_file, $converted_file, $stats_file) = @$info_aref; + + my $selected_entries = "$stats_file.maxC$max_cov.minC$min_cov.maxCV$max_CV.accs"; + my $cmd = "$UTILDIR/nbkc_normalize.pl --stats_file $stats_file " + . " --max_cov $max_cov " + . " --min_cov $MIN_COV " + . " --max_CV $max_CV > $selected_entries"; + + push (@cmds, $cmd) unless (-e "$selected_entries.ok"); + + push (@$info_aref, $selected_entries); + + push (@checkpoints, [$selected_entries, "$selected_entries.ok"]); + + } + + + &process_cmds_parallel(@cmds); ## low memory, all I/O - fine to always run in parallel. + + &process_checkpoints(@checkpoints); + + return; + +} + + +#### +sub run_nkbc_pairs_together { + my ($files_need_stats_aref, $kmer_file, $SS_lib_type, $max_cov, $min_cov, $max_CV) = @_; + + my $left_stats_file = $files_need_stats_aref->[0]->[2]; + my $right_stats_file = $files_need_stats_aref->[1]->[2]; + + my $pair_out_stats_filename = "pairs.K$KMER_SIZE.stats"; + + my $cmd = "$UTILDIR/nbkc_merge_left_right_stats.pl --left $left_stats_file --right $right_stats_file --sorted"; + + $cmd .= " > $pair_out_stats_filename"; + + &process_cmd($cmd) unless (-e "$pair_out_stats_filename.ok"); + my @checkpoints = ( [$pair_out_stats_filename, "$pair_out_stats_filename.ok"] ); + &process_checkpoints(@checkpoints); + + + unless (-s $pair_out_stats_filename) { + die "Error, $pair_out_stats_filename is empty. Be sure to check your fastq reads and ensure that the read names are identical except for the /1 or /2 designation."; + } + + my $selected_entries = "$pair_out_stats_filename.C$max_cov.maxCV$max_CV.accs"; + $cmd = "$UTILDIR/nbkc_normalize.pl --stats_file $pair_out_stats_filename --max_cov $max_cov " + . " --min_cov $MIN_COV --max_CV $max_CV > $selected_entries"; + &process_cmd($cmd) unless (-e "$selected_entries.ok"); + @checkpoints = ( [$selected_entries, "$selected_entries.ok"] ); + &process_checkpoints(@checkpoints); + + push (@{$files_need_stats_aref->[0]}, $selected_entries); + push (@{$files_need_stats_aref->[1]}, $selected_entries); + + + return; + +} + + + +#### +sub process_cmds_parallel { + my @cmds = @_; + + + my @threads; + foreach my $cmd (@cmds) { + # should only be 2 cmds max + my $thread = threads->create('process_cmd', $cmd); + push (@threads, $thread); + } + + my $ret = 0; + + foreach my $thread (@threads) { + $thread->join(); + if (my $error = $thread->error()) { + print STDERR "Error, thread exited with error $error\n"; + $ret++; + } + } + if ($ret) { + die "Error, $ret threads errored out"; + } + + return; +} + +#### +sub process_cmds_serial { + my @cmds = @_; + + foreach my $cmd (@cmds) { + &process_cmd($cmd); + } + + return; +} + + + +#### +sub add_fifo_for_gzip { + my @files = @_; + + foreach my $file (@files) { + if (ref $file eq "ARRAY") { + my @f = &add_fifo_for_gzip(@{$file}); + $file = [@f]; + } + elsif ($file =~ /\.gz$/) { + $file = "<(gunzip -c $file)"; + } elsif ($file =~ /\.xz$/) { + $file = "<(xz -d -c ${file})"; + } elsif ($file =~ /\.bz2$/) { + $file = "<(bunzip2 -dc ${file})"; + } + } + + return(@files); + +} diff --git a/99.scripts/trinity_utils/util/misc/Artemis/join_multi_wig_to_graph_plot.pl b/99.scripts/trinity_utils/util/misc/Artemis/join_multi_wig_to_graph_plot.pl new file mode 100644 index 0000000..b9b9b60 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/Artemis/join_multi_wig_to_graph_plot.pl @@ -0,0 +1,77 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use WigParser; + + +my $usage = "usage: $0 wigA [ wigB ... ]\n\n"; +my @wig_files = @ARGV; + + +unless (@wig_files) { + die $usage; +} + + +main: { + + my @data_arefs; + + my $max_pos = 0; + foreach my $wig_file (@wig_files) { + + my %scaff_to_data = &WigParser::parse_wig($wig_file); + + my @scaffs = keys %scaff_to_data; + if (scalar(@scaffs) != 1) { + die "Error, only a single scaffold can be leveraged by Artemis, and " . scalar(@scaffs) . " scaffs worth of data are found in file: $wig_file:\n@scaffs"; + } + + my $scaff = $scaffs[0]; + + my $data_aref = $scaff_to_data{$scaff}; + + push (@data_arefs, $data_aref); + + my $max = $#$data_aref; + if ($max > $max_pos) { + $max_pos = $max; + } + + + } + + ## output results + + + + + for (my $i = 1; $i <= $max_pos; $i++) { + + my $printed_flag = 0; + foreach my $data_aref (@data_arefs) { + + my $val = $data_aref->[$i] || 0; + if ($printed_flag) { + print "\t"; + } + + print "$val"; + $printed_flag = 1; + + } + print "\n"; + } + + + + exit(0); + + + + +} diff --git a/99.scripts/trinity_utils/util/misc/ButterflyFastaToGraphDot.pl b/99.scripts/trinity_utils/util/misc/ButterflyFastaToGraphDot.pl new file mode 100644 index 0000000..a25008e --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/ButterflyFastaToGraphDot.pl @@ -0,0 +1,78 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib", "$FindBin::RealBin/../../PerlLib/KmerGraphLib"); + +use Fasta_reader; +use StringGraph; +use StringNode; +use ColorGradient; + +my $usage = "usage: $0 Butterfly.fasta\n\n"; + +my $butterfly_fasta_file = $ARGV[0] or die $usage; + + +main: { + + my @seqs_n_paths; + + { + open (my $fh, $butterfly_fasta_file) or die $!; + while (<$fh>) { + if (/^>/) { + /^>(\S+) .* path=\[(.*)\]/ or die "Error, cannot parse header: $_"; + + my $acc = $1; + my $path = $2; + push (@seqs_n_paths, [$acc, $path]); + } + } + close $fh; + + } + + my $graph = new StringGraph(); + + my @colors = &ColorGradient::get_RGB_gradient(scalar(@seqs_n_paths)); + @colors = &ColorGradient::convert_RGB_hex(@colors); + + foreach my $seq_n_path (@seqs_n_paths) { + my ($acc, $path_text) = @$seq_n_path; + + my @path_node_names = &get_path_node_names($path_text); + + my $color = shift @colors; + + $graph->add_sequence_to_graph($acc, \@path_node_names, 1, $color); + + } + + print $graph->toGraphViz(); + +} + +#### +sub get_path_node_names { + my ($path_text) = @_; + + my @node_names; + + my @parts = split(/\s+/, $path_text); + foreach my $part (@parts) { + my ($node_name, $coords) = split(/:/, $part); + + my ($lend, $rend) = split(/-/, $coords); + my $length = $rend-$lend + 1; + + $node_name = "${node_name}_len$length"; + + push (@node_names, $node_name); + } + + return(@node_names); +} + diff --git a/99.scripts/trinity_utils/util/misc/HiCpipe_nameSortedSam_to_raw.pl b/99.scripts/trinity_utils/util/misc/HiCpipe_nameSortedSam_to_raw.pl new file mode 100644 index 0000000..9948245 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/HiCpipe_nameSortedSam_to_raw.pl @@ -0,0 +1,97 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; + +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; +use Data::Dumper; + + +my $usage = "usage: $0 file.sam\n\n"; + +my $sam_file = $ARGV[0] or die $usage; + +main: { + + + + my $sam_reader = new SAM_reader($sam_file); + + print "chr1\tcoord1\tstrand1\tchr2\tcoord2\tstrand2\n"; + + my @entries; + my $curr_read_name = ""; + + while (my $sam_entry = $sam_reader->get_next()) { + my $core_read_name = $sam_entry->get_core_read_name(); + if ($core_read_name ne $curr_read_name) { + if (scalar(@entries) > 1) { + &process_entries(@entries); + } + @entries = (); + + } + $curr_read_name = $core_read_name; + push (@entries, $sam_entry); + } + + # get last ones + if (scalar(@entries) > 1) { + &process_entries(@entries); + + } + + exit(0); +} + +#### +sub process_entries { + my @entries = @_; + + my @left_entries; + my @right_entries; + + foreach my $entry (@entries) { + + if ($entry->get_scaffold_name() eq "*") { next; } # unaligned read + + if ($entry->is_first_in_pair()) { + push (@left_entries, $entry); + } + elsif ($entry->is_second_in_pair()) { + push (@right_entries, $entry); + } + else { + die "Error, not left or right seq of pair: " . Dumper($entry); + } + } + + if (scalar(@left_entries) > 1 || scalar(@right_entries) > 1) { + print STDERR "Error, should only have single read mappings, but have more than one:" + . Dumper(\@left_entries) . Dumper(\@right_entries); + + return; + } + + if (scalar(@left_entries) == 1 && scalar(@right_entries) == 1) { + ## all good. + + my $left_entry = $left_entries[0]; + my $right_entry = $right_entries[0]; + + # "chr1\tcoord1\tstrand1\tchr2\tcoord2\tstrand2\n"; + + print join("\t", + $left_entry->get_scaffold_name(), $left_entry->get_scaffold_position(), $left_entry->get_query_strand(), + $right_entry->get_scaffold_name(), $right_entry->get_scaffold_position(), $right_entry->get_query_strand()) + . "\n"; + + } + + + return; +} diff --git a/99.scripts/trinity_utils/util/misc/Monarch b/99.scripts/trinity_utils/util/misc/Monarch new file mode 100644 index 0000000..f5aa833 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/Monarch @@ -0,0 +1,383 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib", "$FindBin::RealBin/../../PerlLib/KmerGraphLib"); + +use Fasta_reader; + +use ColorGradient; + +use KmerNode; +use KmerGraph; +use Carp; + +use Getopt::Long qw(:config no_ignore_case bundling); + +no warnings 'recursion'; + +my $usage = <<_EOUSAGE_; + +##################################################################################### +# +# Required: +# +# --reads fasta files containing reads (comma-separated list of files: fileA.reads,fileB.reads,... +# basic black coloring in graph.) +# and/or +# +# --misc_seqs fasta files containing reads (comma-separated list of files: fileA.reads,fileB.reads,...) +# each sequence is colored differently and identified in the graph @ edges +# and: +# --assemble +# and/or +# --graph [filename] dot output file for viewing structure in GraphViz +# +# Optional: +# -K Kmer length (default: 24) +# -R maximum recursive search depth from terminus for extension (default: 5) # note can exponentially increases runtime +# +# +# -L minimum assembled sequence length to report (default: 100) +# -E minimum Kmer seed entropy (default: 1.5; note maximum entropy = 2 for nucleotide sequences) +# --min_coverage minimum kmer coverage to report kmer from reads in graph. +# +# +# -v verbose +# +# --no_compact do NOT compact the graph +# +# --min_edge_threshold default(0.1) +# --min_leaf_node_length default(76) +# --min_leaf_node_avg_cov default(1.2) +# +# -N max assemblies to report (default: infinity) +# +##################################################################################### + +_EOUSAGE_ + + ; + + + +my $help_flag; + +my $fasta_file; +my $KmerLength = 24; + +my $max_recurse_depth = 5; + +my $min_length = 100; +my $min_cov = 1; + +our $VERBOSE =0; + +my $MIN_KMER_ENTROPY = 1.5; +my $graph_file; +my $assemble_flag; +my $cds_file; +my $iworm_file; + +my $MIN_EDGE_THRESHOLD = 0.1; +my $MIN_LEAF_NODE_LENGTH = 76; +my $MIN_LEAF_NODE_AVG_COV = 1.2; + + +my $NO_COMPACT = 0; +my $max_asmbls_report = -1; + +my $reads_files; +my $misc_seqs_files; + +&GetOptions ( 'h' => \$help_flag, + + ## required + 'misc_seqs=s' => \$misc_seqs_files, + 'reads=s' => \$reads_files, + + 'K=i' => \$KmerLength, + 'R=i' => \$max_recurse_depth, + + ## options + 'L=i' => \$min_length, + 'E=f' => \$MIN_KMER_ENTROPY, + "min_coverage=i" => \$min_cov, + "graph=s" => \$graph_file, + "assemble" => \$assemble_flag, + "iworm=s" => \$iworm_file, + + "CDS=s" => \$cds_file, + + "min_edge_threshold=f" => \$MIN_EDGE_THRESHOLD, + "min_leaf_node_length=i" => \$MIN_LEAF_NODE_LENGTH, + "min_leaf_node_avg_cov=f" => \$MIN_LEAF_NODE_AVG_COV, + + "no_compact" => \$NO_COMPACT, + 'N=i' => \$max_asmbls_report, + + # hidden option for debugging + 'v' => \$VERBOSE, + + + ); + +if ($help_flag) { + die $usage; +} + +unless ( ($reads_files || $misc_seqs_files) && ($graph_file || $assemble_flag)) { + die $usage; +} + + +our $SEE = $VERBOSE; + + +main: { + + my $KmerGraph = KmerGraph->new($KmerLength); + + if ($reads_files) { + my $seq_counter = 0; + my @files = split(/,/, $reads_files); + foreach my $file (@files) { + my $fasta_reader = new Fasta_reader($file); + + while (my $seq_obj = $fasta_reader->next()) { + my $acc = $seq_obj->get_accession(); + my $sequence = $seq_obj->get_sequence(); + + $seq_counter++; + + print STDERR "\r-parsing seq($seq_counter) "; + + $KmerGraph->add_sequence_to_graph($acc, $sequence, 1, 'black'); + + } + } + } + + if ($misc_seqs_files) { + + my @seqs; + + my @files = split(/,/, $misc_seqs_files); + foreach my $file (@files) { + my $fr = new Fasta_reader($file); + + + + while (my $seq_obj = $fr->next()) { + my $seq = $seq_obj->get_sequence(); + my $acc = $seq_obj->get_accession(); + + if (length($seq) >= $min_length) { + push (@seqs, [$acc, $seq]); + } + + } + } + + my @colors; + if (scalar @seqs > 1) { + @colors = &ColorGradient::get_RGB_gradient(scalar @seqs); + + @colors = &ColorGradient::convert_RGB_hex(@colors); + } + else { + @colors = ('green'); + } + + while (@seqs) { + my $seq_info = shift @seqs; + my ($acc, $seq) = @$seq_info; + my $color = shift @colors; + print STDERR "Adding additional seq trace: $acc, $color\n"; + $KmerGraph->add_sequence_to_graph($acc, $seq, 0, $color); + + } + } + + + + if ($min_cov > 1) { + print STDERR "\n-removing kmer nodes below coverage: $min_cov\n"; + + $KmerGraph->purge_nodes_below_count($min_cov); + } + + + + print STDERR "\nDone adding sequences\n"; + + + + unless ($NO_COMPACT) { + + my $round = 0; + + my ($compacted_graph_flag, $number_dangling_nodes_pruned, $number_pruned_edges); + + do { + + + $compacted_graph_flag = $KmerGraph->compact_graph(); + + $number_dangling_nodes_pruned = $KmerGraph->prune_dangling_nodes(min_leaf_node_length => $MIN_LEAF_NODE_LENGTH, + min_leaf_node_avg_cov => $MIN_LEAF_NODE_AVG_COV); + + $number_pruned_edges = $KmerGraph->prune_low_weight_edges(edge_weight_threshold => $MIN_EDGE_THRESHOLD); + + $round++; + + # no op + print STDERR "Cycling through graph compaction, round: $round\n"; + print STDERR "\tnodes_pruned: $number_dangling_nodes_pruned\n" + . "\tedges_pruned: $number_pruned_edges\n"; + + + } while ($compacted_graph_flag && ($number_dangling_nodes_pruned || $number_pruned_edges)); + + + + } + + print STDERR "done.\n"; + + + if ($assemble_flag) { + + my @path = $KmerGraph->score_nodes(); + + my $assembled_seq = $KmerGraph->extract_sequence_from_path(@path); + + unless ($assembled_seq) { + confess "Error, no assembled sequence form path: @path"; + } + + + my $score = &sum_node_counts(@path); + + my $counter = 0; + if (length($assembled_seq) >= $min_length) { + $counter++; + + my $length_normalized_score = sprintf("%.2f", $score / length($assembled_seq)); + + print ">assembly_$counter;$score;$length_normalized_score len: " . length($assembled_seq) . "\n$assembled_seq\n"; + } + + + while (! $KmerGraph->all_nodes_visited()) { + + if ($max_asmbls_report > 0 && $counter >= $max_asmbls_report) { + last; + } + + my @path = $KmerGraph->extract_path_from_unvisited_node(@path); + + #$KmerGraph->print_path(@path); + + #foreach my $node (@path) { + # print $node->toString(); + #} + + my $assembled_seq = $KmerGraph->extract_sequence_from_path(@path); # marks as visited. + + unless ($assembled_seq) { + confess "Error, no sequence extracted from path: @path"; + } + + if ( length($assembled_seq) >= $min_length) { + $counter++; + + my $score = &sum_node_counts(@path); + my $length_normalized_score = sprintf("%.2f", $score / length($assembled_seq)); + + print ">assembly_$counter;$score;$length_normalized_score len: " . length($assembled_seq) . "\n$assembled_seq\n"; + } + } + } + + ## Write graph file in dot format for GraphViz + if ($graph_file) { + open (my $ofh, ">$graph_file") or die "Error, cannot write graph file to $graph_file "; + print $ofh $KmerGraph->toGraphViz( no_short_singletons => 100); + close $ofh; + } + + + + exit(0); +} + + +#### +sub sum_node_counts { + my @path = @_; + + my $sum = 0; + + foreach my $node (@path) { + my $count = $node->get_count(); + $sum += $count; + } + + return($sum); +} + + +#### +sub compute_entropy { + my ($string) = @_; + + my @chars = split(//, $string); + + my %char_counter; + foreach my $char (@chars) { + $char_counter{$char}++; + } + + my $entropy = 0; + + my $num_chars = length($string); + foreach my $char (keys %char_counter) { + my $count = $char_counter{$char}; + my $prob = $count/$num_chars; + + my $val = $prob * log(1/$prob)/log(2); + $entropy += $val; + } + + return($entropy); +} + + +#### +sub get_highest_scoring_iworm_assembly { + my ($iworm_file) = @_; + + my @entries; + + my $fasta_reader = new Fasta_reader($iworm_file); + while (my $seq_obj = $fasta_reader->next()) { + my $seq = $seq_obj->get_sequence(); + my $acc = $seq_obj->get_accession(); + + my @parts = split(/;/, $acc); + my $cov = pop @parts; + + my $score = $cov * length($seq); + + push (@entries, [$score, $seq]); + } + + @entries = reverse sort {$a->[0]<=>$b->[0]} @entries; + + my $best_entry = shift @entries; + + return($best_entry->[1]); +} diff --git a/99.scripts/trinity_utils/util/misc/Monarch_util/generate_gene_alt_splicing_graphs.pl b/99.scripts/trinity_utils/util/misc/Monarch_util/generate_gene_alt_splicing_graphs.pl new file mode 100644 index 0000000..1db8648 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/Monarch_util/generate_gene_alt_splicing_graphs.pl @@ -0,0 +1,99 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use lib ($ENV{EUK_MODULES}); +use Fasta_reader; + + +my $usage = "usage: $0 transcripts.cdna.fasta\n\n"; + +my $transcripts_fasta_file = $ARGV[0] or die $usage; + +my $graphs_per_dir = 100; + + +main: { + + my $fasta_reader = new Fasta_reader($transcripts_fasta_file) or die $!; + + + my $count = 0; + + my %gene_to_seq; + + while (my $seq_obj = $fasta_reader->next()) { + + my $accession = $seq_obj->get_accession(); + + my $sequence = $seq_obj->get_sequence(); + + my ($trans, $gene) = split(/;/, $accession); + + + $gene_to_seq{$gene}->{$trans} = $sequence; + + } + + foreach my $gene (keys %gene_to_seq) { + + my $trans_href = $gene_to_seq{$gene}; + + my @trans = keys %$trans_href; + unless (scalar @trans > 1) { + # want just alt-splice ones. + next; + } + + $gene =~ s/\W/_/g; + + + my $dir_no = int($count/$graphs_per_dir); + my $outdir = "gene_altSplice_graphs/g_$dir_no"; + if (! -d $outdir) { + &process_cmd("mkdir -p $outdir"); + } + + my $fa_file = "$outdir/$gene.fa"; + open (my $ofh, ">$fa_file") or die $!; + foreach my $trans_acc (keys %$trans_href) { + + my $seq = $trans_href->{$trans_acc}; + print $ofh ">$trans_acc\n$seq\n"; + + } + + + close $ofh; + + my $cmd = "~/SVN/trinityrnaseq/trunk/util/misc/Monarch --misc_seqs $fa_file --graph $fa_file.dot"; + &process_cmd($cmd); + + + $count++; + } + + + print STDERR "\n\nDone.\n\n"; + + exit(0); +} + + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; + + +} diff --git a/99.scripts/trinity_utils/util/misc/Monarch_util/generate_trans_graphs.pl b/99.scripts/trinity_utils/util/misc/Monarch_util/generate_trans_graphs.pl new file mode 100644 index 0000000..68b2bd5 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/Monarch_util/generate_trans_graphs.pl @@ -0,0 +1,71 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use lib ($ENV{EUK_MODULES}); +use Fasta_reader; + +my $usage = "usage: $0 transcripts.cdna.fasta\n\n"; + +my $transcripts_fasta = $ARGV[0] or die $usage; + + +my $graphs_per_dir = 100; + + +main: { + + my $fasta_reader = new Fasta_reader($transcripts_fasta) or die $!; + + + my $count = 0; + + while (my $seq_obj = $fasta_reader->next()) { + + my $accession = $seq_obj->get_accession(); + $accession =~ s/\W/_/g; + + my $sequence = $seq_obj->get_sequence(); + + my $dir_no = int($count/$graphs_per_dir); + my $outdir = "trans_graphs/g_$dir_no"; + if (! -d $outdir) { + &process_cmd("mkdir -p $outdir"); + } + + my $fa_file = "$outdir/$accession.fa"; + open (my $ofh, ">$fa_file") or die $!; + print $ofh ">$accession\n$sequence\n"; + close $ofh; + + my $cmd = "~/SVN/trinityrnaseq/trunk/util/misc/Monarch --misc_seqs $fa_file --graph $fa_file.dot"; + &process_cmd($cmd); + + + $count++; + } + + + print STDERR "\n\nDone.\n\n"; + + exit(0); +} + + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; + + +} diff --git a/99.scripts/trinity_utils/util/misc/N50.pl b/99.scripts/trinity_utils/util/misc/N50.pl new file mode 100644 index 0000000..9d23888 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/N50.pl @@ -0,0 +1,48 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + + +my $usage = "\n\nusage: $0 transcripts.fasta\n\n"; + +my $fasta_file = $ARGV[0] or die $usage; + +main: { + + my $fasta_reader = new Fasta_reader($fasta_file); + + my @seq_lengths; + my $cum_seq_len = 0; + + while (my $seq_obj = $fasta_reader->next()) { + + my $sequence = $seq_obj->get_sequence(); + + my $seq_len = length($sequence); + + $cum_seq_len += $seq_len; + push (@seq_lengths, $seq_len); + } + + @seq_lengths = reverse sort {$a<=>$b} @seq_lengths; + + my $half_cum_len = $cum_seq_len / 2; + + my $partial_sum_len = 0; + foreach my $len (@seq_lengths) { + $partial_sum_len += $len; + + if ($partial_sum_len >= $half_cum_len) { + print "N50: $len\n"; + last; + } + } + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/misc/PerlLib/SegmentGraph.pm b/99.scripts/trinity_utils/util/misc/PerlLib/SegmentGraph.pm new file mode 100644 index 0000000..e8ef743 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/PerlLib/SegmentGraph.pm @@ -0,0 +1,476 @@ +package SegmentGraph; + +use strict; +use warnings; +use Carp; +use Data::Dumper; + +sub new { + my $packagename = shift; + + + my $self = { id_to_node => {}, # primary container for all graph elements + + _next_edges => {}, # id => { id => 1 } + _prev_edges => {}, + }; + + bless ($self, $packagename); + + return($self); +} + +#### +sub delete_node { + my $self = shift; + my ($node) = @_; + + my $node_ID = $node->get_ID(); + + my @next_nodes = $self->get_next_nodes($node); + foreach my $next_node (@next_nodes) { + $self->_delete_prev_edge_to($next_node, $node_ID); + } + my @prev_nodes = $self->get_prev_nodes($node); + foreach my $prev_node (@prev_nodes) { + $self->_delete_next_edge_to($prev_node, $node_ID); + } + + delete( $self->{id_to_node}->{$node_ID} ); + + return; +} + +sub get_next_nodes { + my $self = shift; + my ($node) = @_; + + my $node_ID = $node->get_ID(); + + my @next_nodes; + + if (my $href = $self->{_next_edges}->{$node_ID}) { + my @ids = keys %$href; + foreach my $id (@ids) { + my $next_node = $self->{id_to_node}->{$id} or confess "Error, no node found for ID: $id"; + push (@next_nodes, $next_node); + } + } + + return(@next_nodes); +} + +sub get_prev_nodes { + my $self = shift; + my ($node) = @_; + + my $node_ID = $node->get_ID(); + + my @prev_nodes; + + if (my $href = $self->{_prev_edges}->{$node_ID}) { + my @ids = keys %$href; + foreach my $id (@ids) { + my $next_node = $self->{id_to_node}->{$id} or confess "Error, no node found for ID: $id"; + push (@prev_nodes, $next_node); + } + } + + return(@prev_nodes); +} + + + +sub add_segment { + my $self = shift; + my ($lend, $rend, $parent_feature_name) = @_; + + + + unless ($lend && $rend && $parent_feature_name) { + confess "Error, need params (lend, rend, parent_feature_name)"; + } + + + #print "ADDING SEGMENT: $lend,$rend\n"; + + my $overlapping_node = $self->find_overlapping_segment($lend, $rend); + + if ($overlapping_node) { + + my ($overlapping_node_lend, $overlapping_node_rend) = $overlapping_node->get_coords(); + + #print "Found overlapping nodes: ($lend,$rend) to ($overlapping_node_lend, $overlapping_node_rend)\n"; + + ## number of things could happen here... + + if ($overlapping_node->has_same_coordinates($lend, $rend)) { + $overlapping_node->add_owners($parent_feature_name); + } + elsif ($overlapping_node->overlaps_coordinates($lend, $rend)) { + + ## fracture into segments based on overlaps + + my @coords = sort {$a<=>$b} ($lend, $rend, $overlapping_node->get_coords()); + + ## bounds stay the same, but internal set needs to be adjusted. + my @fractured_segments = ( [$coords[0], $coords[1]-1], + [$coords[1], $coords[2]], + [$coords[2]+1, $coords[3]] ); + + my @remaining_input_segments; + my @contained_by_both; + my @contained_by_overlapping_segment_only; + foreach my $fractured_segment (@fractured_segments) { + my ($seg_lend, $seg_rend) = @$fractured_segment; + + if ($seg_lend > $seg_rend) { next; } + + #print "Searching fragment: $seg_lend,$seg_rend\n"; + + # corresponds to the input and the overlapping segments + if ($overlapping_node->envelops_coordinates($seg_lend, $seg_rend) + && + ($seg_lend >= $lend && $seg_rend <= $rend) + ) { + #print "fragment contained by segment: $seg_lend, $seg_rend\n"; + push (@contained_by_both, $fractured_segment); + } + + elsif ($overlapping_node->envelops_coordinates($seg_lend, $seg_rend)) { + ## a part of the original overlapping segment + push (@contained_by_overlapping_segment_only, $fractured_segment); + } + + + # just the input segment + elsif ($seg_lend >= $lend && $seg_rend <= $rend) { + + #print "adding to remaining segment: $seg_lend, $seg_rend\n"; + push (@remaining_input_segments, $fractured_segment); + } + else { + confess "Error, ended up with coordinate segment that isnt placed: $seg_lend,$seg_rend "; + } + } + + my @owners = $overlapping_node->get_owners(); + + $self->delete_node($overlapping_node); + + #print "Contained by both: " . Dumper(\@contained_by_both) . "\n"; + #print "Contained by overlapping segment only: " . Dumper(\@contained_by_overlapping_segment_only). "\n"; + #print "Remaining input segments: " . Dumper(\@remaining_input_segments) . "\n"; + + + foreach my $seg_coords_aref (@contained_by_both) { + my ($seg_lend, $seg_rend) = @$seg_coords_aref; + $self->_add_segment_node($seg_lend, $seg_rend, [@owners, $parent_feature_name]); + } + + foreach my $seg_coords_aref (@contained_by_overlapping_segment_only) { + my ($seg_lend, $seg_rend)= @$seg_coords_aref; + $self->_add_segment_node($seg_lend, $seg_rend, \@owners); + } + + + foreach my $remaining_input_seg_aref (@remaining_input_segments) { + my ($seg_lend, $seg_rend) = @$remaining_input_seg_aref; + $self->add_segment($seg_lend, $seg_rend, $parent_feature_name); # recursive call + } + } + } + + else { + ## add it + $self->_add_segment_node($lend, $rend, $parent_feature_name); + } + return; +} + + +#### +sub get_all_nodes { + my $self = shift; + + my $nodes_href = $self->{id_to_node}; + my @nodes = values %$nodes_href; + + @nodes = sort {$a->{lend} <=> $b->{lend} + || + $a->{rend} <=> $b->{rend} } @nodes;; + + + return(@nodes); +} + + +#### +sub toString { + my $self = shift; + + + my $text = ""; + + my @nodes = $self->get_all_nodes(); # already nicely sorted + + foreach my $node (@nodes) { + + my ($seg_lend, $seg_rend) = $node->get_coords(); + my @owners = sort $node->get_owners(); + + $text .= "$seg_lend-$seg_rend\t@owners\n"; + } + + return($text); +} + +#### +sub find_overlapping_segment { + my $self = shift; + my ($lend, $rend) = @_; + + my @nodes = $self->get_all_nodes(); + foreach my $node (@nodes) { + unless (ref $node) { + + confess "Error, got node thats not a ref" . Dumper($node); + } + + if ($node->overlaps_coordinates($lend, $rend)) { + return($node); + } + } + +} + + +#### +sub identify_all_owners { + my $self = shift; + + + my %all_owners; + my @nodes = $self->get_all_nodes(); + foreach my $node (@nodes) { + + my @owners = $node->get_owners(); + foreach my $owner (@owners) { + $all_owners{$owner} = 1; + } + } + + return(keys %all_owners); +} + + + +################### +## Private methods +################## + +#### +sub _add_segment_node { + my $self = shift; + + my ($lend, $rend, $parent_feature_name) = @_; + + my $segment_node = SegmentNode->new($lend, $rend, $parent_feature_name); + + my $segment_node_ID = $segment_node->get_ID(); + + $self->{id_to_node}->{$segment_node_ID} = $segment_node; + + return; +} + +#### +sub _delete_prev_edge_to { + my $self = shift; + my ($node_obj, $prev_node_ID) = @_; + unless (ref $node_obj) { + confess "Error, need node_obj as parameter"; + } + + my $node_ID = $node_obj->get_ID(); + + my $prev_edges_href = $self->{_prev_edges}; + delete($prev_edges_href->{$node_ID}->{$prev_node_ID}); + unless (%{$prev_edges_href->{$node_id}}) { + delete($prev_edges_href->{$node_id}); + } + + return; +} + +#### +sub _delete_next_edge_to { + my $self = shift; + my ($node_obj, $next_node_ID) = @_; + unless (ref $node_obj) { + confess "Error, need node_obj as parameter"; + } + + my $node_ID = $node_obj->get_ID(); + + my $next_edges_href = $self->{_next_edges}; + delete($next_edges_href->{$node_ID}->{$next_node_ID}); + unless (%{$next_edges_href->{$node_id}}) { + delete($next_edges_href->{$node_id}); + } + + return; +} + + +##################################################################################################### +package SegmentNode; + +use strict; +use warnings; +use Carp; + +my $NODE_COUNTER = 0; + +sub new { + my $packagename = shift; + my ($lend, $rend, $parent_feature_name) = @_; + + unless ($lend && $rend && $parent_feature_name) { + die "Error, need params(lend, rend, parent_feature_name)"; + } + + my $self = { lend => $lend, + rend => $rend, + owners => {}, + ID => ++$NODE_COUNTER, + }; + + bless ($self, $packagename); + + my @owners; + + if (ref $parent_feature_name eq "ARRAY") { + @owners = @$parent_feature_name; + } + else { + @owners = ($parent_feature_name); + } + + $self->add_owners(@owners); + + return($self); + +} + + +#### +sub get_ID { + my $self = shift; + + return($self->{ID}); +} + +#### +sub has_same_coordinates { + my $self = shift; + my ($lend, $rend) = @_; + + if ($self->{lend} == $lend && $self->{rend} == $rend) { + return(1); + } + else { + return(0); + } +} + +#### +sub add_owners { + my $self = shift; + my (@parent_feature_names) = @_; + + foreach my $parent_feature_name (@parent_feature_names) { + $self->{owners}->{$parent_feature_name} = 1; + } + + return; +} + +#### +sub get_owners { + my $self = shift; + return(keys %{$self->{owners}}); +} + +#### +sub is_contained_by_coordinates { + my $self = shift; + my ($lend, $rend) = @_; + + if ($self->{lend} >= $lend && $self->{rend} <= $rend) { + return(1); + } + else { + return(0); + } +} + +#### +sub envelops_coordinates { + my $self = shift; + my ($lend, $rend) = @_; + + if ($self->{lend} <= $lend && $self->{rend} >= $rend) { + return(1); + } + else { + return(0); + } +} + +#### +sub overlaps_coordinates { + my $self = shift; + my ($lend, $rend) = @_; + + if ($lend <= $self->{rend} && $rend >= $self->{lend}) { + return(1); + } + else { + return(0); + } +} + +#### +sub get_coords { + my $self = shift; + + return($self->{lend}, $self->{rend}); + +} + + +#### +sub has_owners { + my $self = shift; + my @accs = @_; + + foreach my $acc (@accs) { + + if (! exists $self->{owners}->{$acc}) { + return(0); + } + } + + return(1); # must have had them all +} + + + +1; #EOM + + + + + + diff --git a/99.scripts/trinity_utils/util/misc/SAM_coordsorted_max_reads_per_position.pl b/99.scripts/trinity_utils/util/misc/SAM_coordsorted_max_reads_per_position.pl new file mode 100644 index 0000000..8a5811a --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/SAM_coordsorted_max_reads_per_position.pl @@ -0,0 +1,61 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + + +my $usage = "usage: $0 coordSorted.sam max_per_position\n\n"; + +my $sam_file = $ARGV[0] or die $usage; +my $max_per_pos = $ARGV[1] or die $usage; + +main: { + + my $sam_reader = new SAM_reader($sam_file); + + + my $curr_scaffold = ""; + my $curr_pos = -1; + + my $counter = 0; + + while (my $sam_entry = $sam_reader->get_next()) { + + + if ($sam_entry->is_query_unmapped()) { + next; + } + + my $scaff = $sam_entry->get_scaffold_name(); + my $pos = $sam_entry->get_scaffold_position(); + + if ($curr_scaffold ne $curr_scaffold || $curr_pos != $pos) { + + $counter = 0; # re init + $curr_scaffold = $scaff; + $curr_pos = $pos; + + + } + + $counter++; + + if ($counter <= $max_per_pos) { + + print $sam_entry->toString() . "\n"; + } + + } + + + exit(0); +} + + + + diff --git a/99.scripts/trinity_utils/util/misc/SAM_intron_extractor.pl b/99.scripts/trinity_utils/util/misc/SAM_intron_extractor.pl new file mode 100644 index 0000000..331f1ab --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/SAM_intron_extractor.pl @@ -0,0 +1,130 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use SAM_reader; +use SAM_entry; +use Fasta_reader; + +my $usage = "usage: $0 coord_sorted.sam genome.fa [SS_lib_type]\n\n"; + +my $sam_file = $ARGV[0] or die $usage; +my $genome_fa = $ARGV[1] or die $usage; +my $SS_lib_type = $ARGV[2]; + +main: { + + my $sam_reader = new SAM_reader($sam_file); + + my $fasta_reader = new Fasta_reader($genome_fa); + my %genome_seqs = $fasta_reader->retrieve_all_seqs_hash(); + + + my $genome_acc = ""; + my $genome_seq = ""; + + + + while ($sam_reader->has_next()) { + + my $num_introns = 0; + + my $sam_entry = $sam_reader->get_next(); + + my $read_name = $sam_entry->reconstruct_full_read_name(); + + my $cigar = $sam_entry->get_cigar_alignment(); + my $scaffold = $sam_entry->get_scaffold_name(); + + if ($scaffold eq "*") { + # unaligned read + next; + } + + + #print "$scaffold\t$cigar\n"; + + unless ($genome_acc eq $scaffold) { + $genome_acc = $scaffold; + $genome_seq = $genome_seq = $genome_seqs{$genome_acc} or die "Error, no genome seq for acc: $genome_acc"; + } + + my ($genome_coords_aref, $query_coords_aref) = $sam_entry->get_alignment_coords(); + + my $strand = $sam_entry->get_query_strand(); + + my $transcribed_strand = "?"; + if ($SS_lib_type) { + $transcribed_strand = $sam_entry->get_query_transcribed_strand($SS_lib_type); + } + + my $report_txt = ""; + + if (scalar @$genome_coords_aref > 1) { + + my @genome_coords = @$genome_coords_aref; + ## get the introns: + + for (my $i = 1; $i <= $#genome_coords; $i++) { + my $prev_coordset = $genome_coords[$i-1]; + my $curr_coordset = $genome_coords[$i]; + + my ($a_lend, $a_rend) = @$prev_coordset; + my ($b_lend, $b_rend) = @$curr_coordset; + + my $intron_lend = $a_rend + 1; + my $intron_rend = $b_lend - 1; + + my $intron_length = $b_lend - $a_rend - 1; + if ($intron_length < 0) { + die "Error, intron length invalid"; + } + + if ($intron_length >= 20) { + my $intron_seq = substr($genome_seq, $a_rend +1 -1, $intron_length); + + my $left_intron = substr($intron_seq, 0, 2); + my $right_intron = substr($intron_seq, -2); + $num_introns++; + + my $intron_dinucs = "$left_intron..$right_intron"; + if ($transcribed_strand eq "?") { + + ## make an educated guess + if ($intron_dinucs eq "GT..AG" + || + $intron_dinucs eq "GC..AG" + || + $intron_dinucs eq "AT..AC") { + $transcribed_strand = "+"; + } + elsif ($intron_dinucs eq "CT..AC" + || + $intron_dinucs eq "CT..GC" + || + $intron_dinucs eq "GT..AT") { + $transcribed_strand = '-'; + } + } + + + $report_txt .= join("\t", $read_name, + "$scaffold", + "$intron_lend-$intron_rend", + "$transcribed_strand", + "$intron_dinucs") . "\n"; + } + } + if ($num_introns >= 1) { + print "$report_txt\n"; # spacer between SAM entries + } + } + } + + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/misc/SAM_nameSorted_to_uniq_count_stats.pl b/99.scripts/trinity_utils/util/misc/SAM_nameSorted_to_uniq_count_stats.pl new file mode 100644 index 0000000..c3d34ee --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/SAM_nameSorted_to_uniq_count_stats.pl @@ -0,0 +1,276 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "\n\nusage: name_sorted_paired_reads.sam [debug]\n\n"; + +my $sam_file = $ARGV[0] or die $usage; +my $DEBUG = $ARGV[1] || 0; + +if ($sam_file =~ /coord/i) { + die "Your filename contains the term 'coord' which makes me nervous. This needs to be a name-sorted SAM file. Please rename the file or use the correct version here\n."; +} + + +my $DEBUG_OFH; +if ($DEBUG) { + open ($DEBUG_OFH, ">_debug.nameSorted_frag_classes") or die $!; +} + + +=notes + +Gathers all reads having the same core read name (both left and right reads of PE frags). + +Separates the left and right reads into corresponding lists. + +Examines pairwise comparisons between left and right reads, check for same scaffold and that the right reads mate aligned coordinate matches up with that of the left reads aligned coordinate. + +If not properly paired: + have both left and right reads mapped anywhere: improper pair +Otherwise, left-only or right-only counts. + + +=cut + + +main: { + + + my $prev_read_name = ""; + my $prev_scaff_name = ""; + + my @reads; + + my %counts = ( 'UPP' => 0, # unique proper pairs + 'MPP' => 0, # multi-mapped proper pairs + + 'IP' => 0, # improper pairs + + 'UL' => 0, # left unique reads + 'ML' => 0, # left multi-mapped reads + + 'UR' => 0, # right unique reads + 'MR' => 0, # right multi-mapped reads + + 'US' => 0, # single unique-reads + 'MS' => 0, # single multi-mapped reads + ); + + + my $line_counter = 0; + + my $sam_reader = new SAM_reader($sam_file); + while ($sam_reader->has_next()) { + + $line_counter++; + + my $read = $sam_reader->get_next(); + + if ($read->is_query_unmapped()) { next; } + + + my $scaff_name = $read->get_scaffold_name(); + my $core_read_name = $read->get_core_read_name(); + + if ($core_read_name ne $prev_read_name) { + + if (@reads) { + &process_pairs(\@reads, \%counts); + @reads = (); + } + } + + push (@reads, $read); + + + $prev_read_name = $core_read_name; + $prev_scaff_name = $scaff_name; + + + if ($line_counter % 100000 == 0) { + my $count_print = $line_counter; + $count_print=~ s/(\d)(?=(\d{3})+(\D|$))/$1\,/g; + print STDERR "\r[$count_print] lines read "; + } + + + } + + print STDERR "\n\n"; + + &process_pairs(\@reads, \%counts) if @reads; + + + my $sum_reads = 0; + foreach my $count (values %counts) { + $sum_reads += $count; + } + +=try_mirror_bowtie2_summary + +30575 reads; of these: + 30575 (100.00%) were paired; of these: + 2577 (8.43%) aligned concordantly 0 times + 5766 (18.86%) aligned concordantly exactly 1 time + 22232 (72.71%) aligned concordantly >1 times + ---- + 2577 pairs aligned concordantly 0 times; of these: + 133 (5.16%) aligned discordantly 1 time + ---- + 2444 pairs aligned 0 times concordantly or discordantly; of these: + 4888 mates make up the pairs; of these: + 3588 (73.40%) aligned 0 times + 322 (6.59%) aligned exactly 1 time + 978 (20.01%) aligned >1 times +94.13% overall alignment rate + +=cut + + print "Stats for aligned rna-seq fragments (note, not counting those frags where neither left/right read aligned)\n"; + print "\n\n$sum_reads aligned fragments; of these:\n"; + print " " . ($sum_reads - $counts{US} - $counts{MS}) . " were paired; of these:\n"; + print " " . ($counts{UL} + $counts{ML} + $counts{UR} + $counts{MR} + $counts{IP}) . " aligned concordantly 0 times\n"; + print " " . $counts{UPP} . " aligned concordantly exactly 1 time\n"; + print " " . $counts{MPP} . " aligned concordantly >1 times\n"; + print " ----\n"; + print " " . ($counts{UL} + $counts{ML} + $counts{UR} + $counts{MR} + $counts{IP}) . " pairs aligned concordantly 0 times; of these:\n"; + print " " . $counts{IP} . " aligned as improper pairs\n"; + print " " . ($counts{UL} + $counts{ML} + $counts{UR} + $counts{MR}) . " pairs had only one fragment end align to one or more contigs; of these:\n"; + print " " . ($counts{UL} + $counts{ML}) . " fragments had only the left /1 read aligned; of these:\n"; + print " " . $counts{UL} . " left reads mapped uniquely\n"; + print " " . $counts{ML} . " left reads mapped >1 times\n"; + print " " . ($counts{UR} + $counts{MR}) . " fragments had only the right /2 read aligned; of these:\n"; + print " " . $counts{UR} . " right reads mapped uniquely\n"; + print " " . $counts{MR} . " right reads mapped >1 times\n"; + print "Overall, " . sprintf("%.2f", ($counts{UPP} + $counts{MPP})/$sum_reads * 100) . "% of aligned fragments aligned as proper pairs\n\n\n"; + + if ($counts{US} + $counts{MS}) { + print " " . ($counts{US} + $counts{MS}) . " were single-end reads; of these:\n"; + print " " . $counts{US} . " single-end reads mapped uniquely\n"; + print " " . $counts{MS} . " single-end reads mapped >1 times\n"; + } + + close $DEBUG_OFH if $DEBUG; + + exit(0); + +} + + + +#### +sub process_pairs { + my ($reads_aref, $counts_href) = @_; + + my @reads = @$reads_aref; + + my @left_reads; + my @right_reads; + + + foreach my $read (@reads) { + if ($read->is_first_in_pair()) { + push (@left_reads, $read); + } + elsif ($read->is_second_in_pair()) { + push (@right_reads, $read); + } + } + + my $got_left_read = (@left_reads) ? 1 : 0; + my $got_right_read = (@right_reads) ? 1: 0; + + + ## check to see if we have proper pairs: + my @proper_pairs; + + + pair_search: + foreach my $left_read (@left_reads) { + + unless ($left_read->is_proper_pair()) { next; } + + + my $aligned_pos = $left_read->get_aligned_position(); + + foreach my $right_read (@right_reads) { + + unless ($right_read->is_proper_pair()) { next; } + + if ($left_read->get_scaffold_name() eq $right_read->get_scaffold_name() + && + $right_read->get_mate_scaffold_position() == $aligned_pos) { + + push (@proper_pairs, [$left_read, $right_read]); + + last; + } + } + } + + + my $class = ""; + + if (@proper_pairs) { + + if (scalar(@proper_pairs) == 1) { + $class = "UPP"; + } + else { + # multi ampped proper pairs + $class = "MPP"; + } + } + elsif ($got_left_read && $got_right_read) { + $class = "IP"; + } + elsif ($got_left_read) { + + if (scalar(@left_reads) == 1) { + $class = "UL"; + } + else { + $class = "ML"; + } + } + elsif ($got_right_read) { + + if (scalar(@right_reads) == 1) { + $class = "UR"; + } + else { + $class = "MR"; + } + } + else { + if (scalar(@reads) == 1) { + $class = "US"; + } + else { + $class = "MS"; + } + } + + $counts_href->{$class}++; + + if ($DEBUG) { + my $core_read_name = $reads[0]->get_core_read_name(); + + print $DEBUG_OFH join("\t", $core_read_name, $class) . "\n"; + } + + return; +} + + + + + + diff --git a/99.scripts/trinity_utils/util/misc/SAM_pair_to_bed.pl b/99.scripts/trinity_utils/util/misc/SAM_pair_to_bed.pl new file mode 100644 index 0000000..4c85907 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/SAM_pair_to_bed.pl @@ -0,0 +1,93 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; +use Overlap_piler; + +my $usage = "usage: $0 file.sam\n\n"; + +my $sam_file = $ARGV[0] or die $usage; + +main: { + + my $sam_reader = new SAM_reader($sam_file); + + my %core_to_coords; + + while ($sam_reader->has_next()) { + + my $sam_entry = $sam_reader->get_next(); + + if ($sam_entry->is_query_unmapped()) { + next; + } + + my $read_name = $sam_entry->get_read_name(); + my $scaff_name = $sam_entry->get_scaffold_name(); + + my $strand = $sam_entry->get_query_strand(); + + my ($genome_coords_aref, $query_coords_aref) = $sam_entry->get_alignment_coords(); + + + my $core_scaff = join("$;", $sam_entry->get_core_read_name(), $scaff_name); + + my @coords; + foreach my $segment (@$genome_coords_aref) { + my ($lend, $rend) = @$segment; + + push (@{$core_to_coords{$core_scaff}}, [$lend, $rend]); + } + + } + + foreach my $core_scaff (keys %core_to_coords) { + + my @coords = @{$core_to_coords{$core_scaff}}; + my ($read_core_name, $scaff_name) = split(/$;/, $core_scaff); + + @coords = sort {$a<=>$b} @coords; + @coords = &Overlap_piler::simple_coordsets_collapser(@coords); + + + my $span_lend = $coords[0]->[0]; + my $span_rend = $coords[$#coords]->[1]; + + my @lengths; + my @starts; + my $num_segments = 0; + foreach my $segment (@coords) { + my ($lend, $rend) = @$segment; + + my $length = $rend - $lend + 1; + push (@lengths, $length); + push (@starts, $lend - $span_lend); + $num_segments++; + } + + $span_lend--; # coordinate is zero-based, and rend is exclusive + + print join("\t", + $scaff_name, + $span_lend, + $span_rend, + $read_core_name, + 0, + "+", + $span_lend, + $span_rend, + ".", + $num_segments, + join(",", @lengths), + join(",", @starts), + ) . "\n"; + } + + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/misc/SAM_show_alignment.pl b/99.scripts/trinity_utils/util/misc/SAM_show_alignment.pl new file mode 100644 index 0000000..0870d43 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/SAM_show_alignment.pl @@ -0,0 +1,238 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use lib ($ENV{EUK_MODULES}); +use Fasta_reader; +use SAM_reader; +use SAM_entry; +use Data::Dumper; + +my $FASTA_LENGTH = 60; + +my $usage = "usage: $0 alignments.sam target.fasta [JUST_ALIGN_STATS=0]\n\n"; + +my $sam_alignments = $ARGV[0] or die $usage; +my $target_fasta = $ARGV[1] or die $usage; +my $JUST_STATS = $ARGV[2] || 0; + +main: { + + my $fasta_reader = new Fasta_reader($target_fasta); + my %fasta_seqs = $fasta_reader->retrieve_all_seqs_hash(); + + my $sam_reader = new SAM_reader($sam_alignments); + + my $curr_scaff_acc = ""; + my $curr_scaff_seq = ""; + + while (my $sam_entry = $sam_reader->get_next() ) { + + my $read_name = $sam_entry->reconstruct_full_read_name(); + my $read_seq = uc $sam_entry->get_sequence(); + my $scaff_name = $sam_entry->get_scaffold_name(); + + if ($scaff_name eq "*") { next; } # not aligned. + + + if ($scaff_name ne $curr_scaff_acc) { + $curr_scaff_acc = $scaff_name; + $curr_scaff_seq = uc $fasta_seqs{$scaff_name}; + } + + my ($genome_coords_aref, $read_coords_aref) = $sam_entry->get_alignment_coords(); + + unless (scalar(@$genome_coords_aref) >= 1) { next; } + + unless ($JUST_STATS) { + + ## coordinate dump. + + print "// $scaff_name (top) vs. $read_name (bottom):\n\n"; + + print "$scaff_name length: " . length($curr_scaff_seq) . "\n"; + print "$read_name length: " . length($read_seq) . "\n"; + + print "$scaff_name\t$read_name\n"; + + for (my $i = 0; $i <= $#$genome_coords_aref; $i++) { + my ($genome_lend, $genome_rend) = @{$genome_coords_aref->[$i]}; + my ($read_lend, $read_rend) = @{$read_coords_aref->[$i]}; + print "$genome_lend-$genome_rend\t$read_lend-$read_rend\n"; + } + print "\n"; + } + + + eval { + &draw_alignment($scaff_name, $read_name, $genome_coords_aref, $read_coords_aref, \$curr_scaff_seq, \$read_seq); + }; + if ($@) { + print STDERR "$@\n"; + } + + } + + + exit(0); +} + + + +#### +sub draw_alignment { + my ($scaff_name, $read_name, $genome_coords_aref, $read_coords_aref, $curr_scaff_seq_sref, $read_seq_sref) = @_; + + my $scaff_align_string = ""; + my $read_align_string = ""; + + my $prev_scaff_coord = 0; + my $prev_read_coord = 0; + + my $num_indels = scalar(@$genome_coords_aref) - 1; + my $num_aligned_bases = 0; + + while (@$genome_coords_aref) { + my $genome_coordset = shift @$genome_coords_aref; + my $read_coordset = shift @$read_coords_aref; + + my ($genome_lend, $genome_rend) = @$genome_coordset; + my ($read_lend, $read_rend) = @$read_coordset; + + ## sometimes bwasw is off at the end of the contig + if ($genome_rend > length($$curr_scaff_seq_sref)) { + + my $delta = $genome_rend - length($$curr_scaff_seq_sref); + $genome_rend -= $delta; + $read_rend -= $delta; + # print STDERR "*adjusting by $delta\n"; + } + + $num_aligned_bases += $read_rend - $read_lend + 1; + + if ($prev_read_coord > 0 && $read_lend - $prev_read_coord > 1) { + ## add gaps to read string + for (my $i = $prev_read_coord + 1; $i < $read_lend; $i++) { + $read_align_string .= substr($$read_seq_sref, $i-1, 1); + $scaff_align_string .= "-";; + } + } + + if ($prev_scaff_coord > 0 && $genome_lend - $prev_scaff_coord > 1) { + # add gaps to genome string + for (my $i = $prev_scaff_coord + 1; $i < $genome_lend; $i++) { + $scaff_align_string .= substr($$curr_scaff_seq_sref, $i-1, 1); + $read_align_string .= "-"; + } + } + + $prev_read_coord = $read_rend; + $prev_scaff_coord = $genome_rend; + + + my $read_seq_region_len = $read_rend - $read_lend + 1; + my $genome_seq_region_len = $genome_rend - $genome_lend + 1; + + my $read_seq_region = substr($$read_seq_sref, $read_lend - 1, $read_rend - $read_lend + 1); + my $genome_seq_region = substr($$curr_scaff_seq_sref, $genome_lend - 1, $genome_rend - $genome_lend + 1); + + + + if (length($read_seq_region) != length($genome_seq_region)) { + die "Error, lengths of regions are different: ($read_seq_region_len vs. $genome_seq_region_len) extracted:\n" + . "read_seq_region:\t" . $read_seq_region . "\n" + . "genome_seq_regn:\t" . $genome_seq_region . "\n"; + } + + + $scaff_align_string .= $genome_seq_region; + $read_align_string .= $read_seq_region; + + + + } + + my $alignment_length = length($scaff_align_string); + my $num_gaps = 0; + while ($scaff_align_string =~ /-/g) { $num_gaps++; } + while ($read_align_string =~ /-/g) { $num_gaps++; } + + my $percent_gap = $num_gaps / $alignment_length * 100; + + eval { + my ($num_mismatches) = &print_pretty($scaff_align_string, $read_align_string); + + print join("\t", "#", "scaff_name", "read_name", "read_length", "aligned_bases", "matches", "mismatches", "indel_bkpts", "sum_indel_lens", "pct_mismatches", "pct_indel_bkpts", "pct_indel_lens") . "\n"; + print join("\t", "#", $scaff_name, $read_name, length($$read_seq_sref), + $num_aligned_bases, $num_aligned_bases - $num_mismatches, $num_mismatches, $num_indels, $num_gaps, + sprintf("%.2f", $num_mismatches/$num_aligned_bases*100), + sprintf("%.2f", $num_indels/$num_aligned_bases*100), + sprintf("%.2f", $percent_gap), + ) . "\n\n\n"; + }; + if ($@) { + print STDERR "$@\n"; + die; + } + + return; + + +} + +#### +sub print_pretty { + my ($scaff_align_string, $read_align_string) = @_; + + if (length($scaff_align_string) != length($read_align_string) ) { + die "Error, alignment lengths differ.:\n" + . "Scaff_align_string: " . length($scaff_align_string) . "\t$scaff_align_string\n" + . "Read_align_string: " . length($read_align_string) . "\t$read_align_string\n"; + + } + + my @scaff_align_chars = split(//, $scaff_align_string); + my @read_align_chars = split(//, $read_align_string); + + + my $top_text = ""; + my $match_text = ""; + my $bottom_text = ""; + + my $num_mismatches = 0; + + for (my $i = 0; $i <= $#scaff_align_chars; $i++) { + + if ($i != 0 && $i % 60 == 0) { + + print join("\n", $top_text, $match_text, $bottom_text) . "\n\n" unless $JUST_STATS; + + $top_text = ""; $match_text = ""; $bottom_text = ""; + + } + + my $match_char = "."; + if ($scaff_align_chars[$i] ne "-" && $read_align_chars[$i] ne "-") { + + $match_char = ($scaff_align_chars[$i] eq $read_align_chars[$i]) ? "|" : "*"; + + if ($match_char eq "*") { + $num_mismatches++; + } + + } + $top_text .= $scaff_align_chars[$i]; + $bottom_text .= $read_align_chars[$i]; + $match_text .= $match_char; + + } + + if ($top_text) { + print join("\n", $top_text, $match_text, $bottom_text) . "\n\n" unless $JUST_STATS; + } + + + return ($num_mismatches); + +} diff --git a/99.scripts/trinity_utils/util/misc/SAM_show_alignment.summarize_stats.pl b/99.scripts/trinity_utils/util/misc/SAM_show_alignment.summarize_stats.pl new file mode 100644 index 0000000..0aaa659 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/SAM_show_alignment.summarize_stats.pl @@ -0,0 +1,180 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Statistics::Descriptive; + +my $usage = "usage: $0 bwasw.sam.align_stats [longest_contig_only] [no_require_100pct_align]\n\n"; + + +my $align_stats_file = $ARGV[0] or die $usage; +my $longest_contig_only_flag = $ARGV[1] || 0; +my $no_require_100pct_align = $ARGV[2] || 0; + + +=format + +0 # +1 scaff_name +2 read_name +3 read_length +4 aligned_bases +5 matches +6 mismatches +7 indel_bkpts +8 sum_indel_lens +9 pct_mismatches +10 pct_indel_bkpts +11 pct_indel_lens + + +=cut + + + ; + + +my @pct_mismatches; +my @pct_indels; + +my $sum_length = 0; +my $sum_mismatches = 0; +my $sum_indels = 0; + + +my @perfect_aligns; +my @imperfect_aligns; + +open (my $fh, $align_stats_file) or die $!; + + + +my %core_acc_to_entries; + + +my $prev_read_name = ""; + +while (<$fh>) { + chomp; + my $line = $_; + + my @x = split(/\t/); + if (scalar (@x) == 12 && $x[3] =~ /^\d+$/) { + + my $read_name = $x[2]; + if ($read_name eq $prev_read_name) { + next; # only one alignment per read + } + $prev_read_name = $read_name; + + + + my $core_read_name = $read_name; + $core_read_name =~ s/_\d+$//; + $core_read_name =~ s/\.\d+\.PbioCR$//; + my $read_length = $x[3]; + + push (@{$core_acc_to_entries{$core_read_name}}, { line => $line, + read_len => $read_length, + }); + } +} +close $fh; + + +foreach my $core_read_acc (keys %core_acc_to_entries) { + + my @alignments = @{$core_acc_to_entries{$core_read_acc}}; + + if ($longest_contig_only_flag) { + @alignments = sort {$a->{read_len}<=>$b->{read_len}} @alignments; + my $longest_read = pop @alignments; + @alignments = ($longest_read); + } + + foreach my $alignment (@alignments) { + + + my $line = $alignment->{line}; + my @x = split(/\t/, $line); + + + my $seq_length = $x[3]; + my $aligned_bases = $x[4]; + my $mismatches = $x[6]; + my $indels = $x[8]; + + my $pct_mism = $x[9]; + my $pct_indl = $x[11]; + + + push (@pct_mismatches, $pct_mism); + push (@pct_indels, $pct_indl); + + $sum_length += $aligned_bases; + $sum_mismatches += $mismatches; + $sum_indels += $indels; + + + if ($mismatches == 0 && $indels == 0 && ($no_require_100pct_align || $seq_length == $aligned_bases)) { + push (@perfect_aligns, $line); + } + else { + push (@imperfect_aligns, $line); + } + + } +} + + +my $avg_pct_mismatch = $sum_mismatches / $sum_length * 100; +my $avg_pct_indel = $sum_indels / $sum_length * 100; + +print "// base stats:\n"; +print "Avg_pct_mismatch_all_bases: $avg_pct_mismatch\n"; +print "Avg_pct_indel_all_bases: $avg_pct_indel\n"; + + +print "\n// assembly stats:\n"; +my $stat = Statistics::Descriptive::Sparse->new(); +$stat->add_data(@pct_mismatches); +print "Mean pct_mismatch of assembly: " . $stat->mean() . "\n"; + +$stat->clear(); +$stat->add_data(@pct_indels); +print "Mean pct_indel of assembly: " . $stat->mean() . "\n"; +print "\n\n"; + + + +################################################ +# write some files for downstream analysis +################################################ + + +# write files for interrogation using R +open (my $ofh, ">$align_stats_file.pct_mismatch.dat") or die $!; +print $ofh join("\n", @pct_mismatches) . "\n"; +close $ofh; + +open ($ofh, ">$align_stats_file.pct_indel.dat") or die $!; +print $ofh join("\n", @pct_indels) . "\n"; +close $ofh; + +open ($ofh, ">$align_stats_file.perfect_aligns.txt") or die $!; +print $ofh join("\n", @perfect_aligns) . "\n" if @perfect_aligns; +close $ofh; + + +open ($ofh, ">$align_stats_file.imperfect_aligns.txt") or die $!; +print $ofh join("\n", @imperfect_aligns) . "\n" if @imperfect_aligns; +close $ofh; + +open ($ofh, ">$align_stats_file.all_aligns.txt") or die $!; +print $ofh join("\n", @perfect_aligns, @imperfect_aligns) . "\n" if (@perfect_aligns || @imperfect_aligns); +close $ofh; + + + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/SAM_sortAny_to_count_stats.pl b/99.scripts/trinity_utils/util/misc/SAM_sortAny_to_count_stats.pl new file mode 100644 index 0000000..46c444f --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/SAM_sortAny_to_count_stats.pl @@ -0,0 +1,177 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "\n\nusage: aligned_reads.sam [debug]\n\n" + . "Note: uses the bitflag settings for counting entries. Proper-pairing supercedes other settings.\n\n"; + +my $sam_file = $ARGV[0] or die $usage; +my $DEBUG = $ARGV[1] || 0; + + +my $DEBUG_OFH; +if ($DEBUG) { + open ($DEBUG_OFH, ">_debug.frag_classes") or die $!; +} + +=notes + +Entirely works off flag settings. + +Proper pairs trump individual left/right alignments. + +Improper pairs: left and right read alignments exist anywhere but not flagged as proper pairs. + +=cut + +main: { + + my $prev_read_name = ""; + my $prev_scaff_name = ""; + + my %counts; + + my $count = 0; + + my $sam_reader = new SAM_reader($sam_file); + while ($sam_reader->has_next()) { + + $count++; + + my $read = $sam_reader->get_next(); + + if ($read->is_query_unmapped()) { next; } # ignore unaligned read entries. + + my $scaff_name = $read->get_scaffold_name(); + my $core_read_name = $read->get_core_read_name(); + + if ($read->is_paired()) { + + if ($read->is_proper_pair()) { + ## erase any single entries. + if (exists $counts{ $core_read_name . "::L" }) { + delete $counts{ $core_read_name . "::L" }; + } + if (exists $counts{ $core_read_name . "::R" } ) { + delete $counts{ $core_read_name . "::R" }; + } + $counts{ $core_read_name . "::PP" } = 1; + } + else { + # not propper pair + unless (exists $counts{ $core_read_name . "::PP" }) { + if ($read->is_first_in_pair()) { + $counts{ $core_read_name . "::L" } = 1; + } + elsif ($read->is_second_in_pair()) { + $counts{ $core_read_name . "::R" } = 1; + } + } + } + } + else { + $counts{ $core_read_name . "::S" } = 1; + } + + if ($count % 100000 == 0) { + my $count_print = $count; + $count_print=~ s/(\d)(?=(\d{3})+(\D|$))/$1\,/g; + print STDERR "\r[$count_print] "; + } + } + print STDERR "\n\n"; + + ## identify improper pairs + my @reads = keys %counts; + foreach my $read (@reads) { + if ($read =~ /::L$/) { + my $core_name = $read; + $core_name =~ s/::L$//; + + if (exists($counts{ $core_name . "::R" }) ) { + ## count as improper pair instead + $counts{ $core_name . "::IP" } = 1; + delete $counts{$read}; + delete $counts{ $core_name . "::R" }; + + } + } + } + + + ## sum counts, generate summary + + my $count_PP = 0; + my $count_L = 0; + my $count_R = 0; + my $count_S = 0; + my $count_IP = 0; # improper pairs + + my $total = 0; + + foreach my $read (keys %counts) { + + my @x = split(/::/, $read); + my $class = pop @x; + my $core_read_name = join("::", @x); + + if ($class eq "PP") { + $count_PP += 2; + $total += 2; + } + elsif ($class eq "IP") { + $count_IP += 2; + $total += 2; + + } + elsif ($class eq "L") { + $count_L++; + $total++; + + } + elsif ($class eq "R") { + $count_R++; + $total++; + + } + elsif ($class eq "S") { + $count_S++; + $total++; + + } + + if ($DEBUG) { + print $DEBUG_OFH join("\t", $core_read_name, $class) . "\n"; + + } + + } + + print "Proper_pair:\t$count_PP\t" . sprintf("%.3f%%", $count_PP/$total*100) . "\n" + . "Improper_pair:\t$count_IP\t" . sprintf("%.3f%%", $count_IP/$total*100) . "\n" + . "Left-only:\t$count_L\t" . sprintf("%.3f%%", $count_L/$total*100) . "\n" + . "Right-only:\t$count_R\t" . sprintf("%.3f%%", $count_R/$total*100) . "\n" + . "Single_end:\t$count_S\t" . sprintf("%.3f%%", $count_S/$total*100) . "\n\n"; + + print "Total aligned frags: $total\n\n"; + + + + close $DEBUG_OFH if $DEBUG; + + exit(0); + + +} + + + + + + diff --git a/99.scripts/trinity_utils/util/misc/SAM_toString.pl b/99.scripts/trinity_utils/util/misc/SAM_toString.pl new file mode 100644 index 0000000..e3f43e4 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/SAM_toString.pl @@ -0,0 +1,50 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "usage: $0 file.sam\n\n"; + +my $sam_file = $ARGV[0] or die $usage; + + +main: { + my $sam_reader = new SAM_reader($sam_file); + + while (my $sam_entry = $sam_reader->get_next()) { + my ($genome_coords_aref, $read_coords_aref) = $sam_entry->get_alignment_coords(); + + my $read_name = $sam_entry->get_read_name(); + my $scaffold = $sam_entry->get_scaffold_name(); + + print join("\t", $scaffold, &get_coord_string($genome_coords_aref), + $read_name, &get_coord_string($read_coords_aref)) . "\n"; + } + + exit(0); +} + + +#### +sub get_coord_string { + my ($coords_aref) = @_; + + my $text = ""; + + foreach my $coordset (@$coords_aref) { + my ($lend, $rend) = @$coordset; + + if ($text) { + $text .= ","; + } + + $text .= "$lend-$rend"; + } + + return($text); +} diff --git a/99.scripts/trinity_utils/util/misc/SAM_to_bed.pl b/99.scripts/trinity_utils/util/misc/SAM_to_bed.pl new file mode 100644 index 0000000..2a41422 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/SAM_to_bed.pl @@ -0,0 +1,77 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "usage: $0 file.sam\n\n"; + +my $sam_file = $ARGV[0] or die $usage; + +main: { + + my $sam_reader = new SAM_reader($sam_file); + + while ($sam_reader->has_next()) { + + my $sam_entry = $sam_reader->get_next(); + + if ($sam_entry->is_query_unmapped()) { + next; + } + + my $read_name = $sam_entry->get_read_name(); + my $scaff_name = $sam_entry->get_scaffold_name(); + + my $strand = $sam_entry->get_query_strand(); + + my ($genome_coords_aref, $query_coords_aref) = $sam_entry->get_alignment_coords(); + + my @coords; + foreach my $segment (@$genome_coords_aref) { + my ($lend, $rend) = @$segment; + + push (@coords, $lend, $rend); + } + + @coords = sort {$a<=>$b} @coords; + my $span_lend = shift @coords; + my $span_rend = pop @coords; + + my @lengths; + my @starts; + my $num_segments = 0; + foreach my $segment (@$genome_coords_aref) { + my ($lend, $rend) = @$segment; + + my $length = $rend - $lend + 1; + push (@lengths, $length); + push (@starts, $lend - $span_lend); + $num_segments++; + } + + $span_lend--; # coordinate is zero-based, and rend is exclusive + + print join("\t", + $scaff_name, + $span_lend, + $span_rend, + $read_name, + 0, + $strand, + $span_lend, + $span_rend, + ".", + $num_segments, + join(",", @lengths), + join(",", @starts), + ) . "\n"; + } + + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/misc/SAM_to_fasta.pl b/99.scripts/trinity_utils/util/misc/SAM_to_fasta.pl new file mode 100644 index 0000000..7a11664 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/SAM_to_fasta.pl @@ -0,0 +1,34 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; +use Nuc_translator; + +my $usage = "usage: $0 file.sam\n\n"; +my $sam_file = $ARGV[0] or die $usage; + +main: { + + + my $sam_reader = new SAM_reader($sam_file); + + while (my $sam_entry = $sam_reader->get_next()) { + + my $read_name = $sam_entry->reconstruct_full_read_name(); + my $sequence = $sam_entry->get_sequence(); + + if ((! $sam_entry->is_query_unmapped()) && $sam_entry->get_query_strand() eq '-') { + $sequence = &reverse_complement($sequence); + } + + print ">$read_name\n$sequence\n"; + } + + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/misc/SAM_to_gff3.minimap2.pl b/99.scripts/trinity_utils/util/misc/SAM_to_gff3.minimap2.pl new file mode 100644 index 0000000..f49ab94 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/SAM_to_gff3.minimap2.pl @@ -0,0 +1,182 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; +use Carp; +use List::Util qw(min max); + +my $usage = "usage: $0 file.sam [debug_flag=0]\n\n"; + +my $sam_file = $ARGV[0] or die $usage; + +my $DEBUG = $ARGV[1]; + +main: { + + my %PATH_COUNTER; + + my $sam_reader = new SAM_reader($sam_file); + + while ($sam_reader->has_next()) { + + my $sam_entry = $sam_reader->get_next(); + + if ($sam_entry->is_query_unmapped()) { + next; + } + + my $read_name = $sam_entry->get_read_name(); + if ($read_name =~ /\.p\d$/) { + # not the first path reported. + next; + } + + my $sequence = $sam_entry->get_sequence(); + if ($sequence eq "*") { + next; + } + + my $sam_line = $sam_entry->get_original_line(); + + if ($DEBUG) { + print "$sam_line\n"; + } + + + my $NM = 0; # full edit distance (includes mismatches and indels) + + if ($sam_line =~ /NM:i:(\d+)/) { + $NM = $1; + } + else { + die "Error, couldn't extract num mismatches from sam line: $sam_line"; + } + my $cigar_align = $sam_entry->get_cigar_alignment(); + my $num_indel_nts = 0; + while($cigar_align =~ /(\d+)[DI]/g) { + $num_indel_nts += $1; + } + + my $num_mismatches = $NM - $num_indel_nts; + + if ($num_mismatches < 0) { + confess "Error, calculated negative mismatch count from: NM:$NM, indel:$num_indel_nts, cigar: $cigar_align"; + } + + + + + my $scaff_name = $sam_entry->get_scaffold_name(); + + my $strand = $sam_entry->get_query_strand(); + + + my ($genome_coords_aref, $query_coords_aref) = $sam_entry->get_alignment_coords(); + + my $min_coord = 0 + 'inf'; + my $max_coord = 0 - 'inf'; + + my $align_len = 0; + { + foreach my $coordset (@$genome_coords_aref) { + my $seglen = abs($coordset->[1] - $coordset->[0]) + 1; + $align_len += $seglen; + + $min_coord = min($min_coord, $coordset->[0], $coordset->[1]); + $max_coord = max($max_coord, $coordset->[0], $coordset->[1]); + + if ($DEBUG) { + print STDERR join("\t", $coordset->[0], $coordset->[1], "seglen: $seglen") . "\n"; + } + } + } + + + + if ($DEBUG) { + print STDERR "num_mismatches: $num_mismatches, align_length: $align_len\n"; + } + + my $per_id = sprintf("%.1f", 100 - $num_mismatches/$align_len * 100); + + my $align_counter = "$read_name.p" . ++$PATH_COUNTER{$read_name}; + + + my @genome_n_trans_coords; + + + while (@$genome_coords_aref) { + my $genome_coordset_aref = shift @$genome_coords_aref; + my $trans_coordset_aref = shift @$query_coords_aref; + + my ($genome_lend, $genome_rend) = @$genome_coordset_aref; + my ($trans_lend, $trans_rend) = sort {$a<=>$b} @$trans_coordset_aref; + + push (@genome_n_trans_coords, [ $genome_lend, $genome_rend, $trans_lend, $trans_rend ] ); + + } + + ## merge neighboring features if within a short distance unlikely to represent an intron. + my @merged_coords; + push (@merged_coords, shift @genome_n_trans_coords); + + my $MERGE_DIST = 10; + while (@genome_n_trans_coords) { + my $coordset_ref = shift @genome_n_trans_coords; + my $last_coordset_ref = $merged_coords[$#merged_coords]; + + if ($coordset_ref->[0] - $last_coordset_ref->[1] <= $MERGE_DIST) { + # merge it. + $last_coordset_ref->[1] = $coordset_ref->[1]; + + if ($strand eq "+") { + $last_coordset_ref->[3] = $coordset_ref->[3]; + } else { + $last_coordset_ref->[2] = $coordset_ref->[2]; + } + } + else { + # not merging. + push (@merged_coords, $coordset_ref); + } + } + + #my $trans_align_len = 0; + #oreach my $coordset_ref (@merged_coords) { + # my ($genome_lend, $genome_rend, $trans_lend, $trans_rend) = @$coordset_ref; + # $trans_align_len += $trans_rend - $trans_lend + 1; + #} + # + #if ($DEBUG) { + # print "interval-based alignment length: $trans_align_len\n"; + # + + + + foreach my $coordset_ref (@merged_coords) { + my ($genome_lend, $genome_rend, $trans_lend, $trans_rend) = @$coordset_ref; + + print join("\t", + $scaff_name, + "minimap2", + "cDNA_match", + $genome_lend, $genome_rend, + $per_id, + $strand, + ".", + "ID=$align_counter;Parent=$align_counter.mrna;Target=$read_name $trans_lend $trans_rend") . "\n"; + } + print "\n"; + + + + } + + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/misc/SRA_to_fastq.notes b/99.scripts/trinity_utils/util/misc/SRA_to_fastq.notes new file mode 100644 index 0000000..2b53b76 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/SRA_to_fastq.notes @@ -0,0 +1 @@ +SRA_TOOLKIT/fastq-dump --defline-seq '@$sn[_$rn]/$ri' --split-files file.sra diff --git a/99.scripts/trinity_utils/util/misc/SRA_to_fastq.pl b/99.scripts/trinity_utils/util/misc/SRA_to_fastq.pl new file mode 100644 index 0000000..5a2b376 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/SRA_to_fastq.pl @@ -0,0 +1,84 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use File::Basename; + +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + +my $usage = "\n\nusage: $0 --prefix \$prefix fileA.sra [fileB.sra ...]\n\n\n"; + + +my $prefix; + +&GetOptions ('prefix=s' => \$prefix); + +my @sra_files = @ARGV; + +unless (@ARGV) { + die $usage; +} + +unless (@sra_files) { + die $usage; +} + +my $fastq_dump_path = `sh -c "command -v fastq-dump"`; +unless ($fastq_dump_path && $fastq_dump_path =~ /\w/) { + die "Error, cannot find 'fastq-dump' utility in your PATH. Be sure you have SRA toolkit installed and fastq-dump in your PATH setting. "; +} + +my @core_names; +foreach my $sra_file (@sra_files) { + + my $core_name = $sra_file; + $core_name =~ s/\.sra$//; + $core_name = basename($core_name); + push (@core_names, $core_name); + + my $cmd = "fastq-dump --defline-seq '@\$sn[_\$rn]/\$ri' --split-files $sra_file"; + &process_cmd($cmd) unless (-s "${core_name}_1.fastq"); + +} + +if ($prefix) { + + my @final_cmds; + my @tmp_files; + for my $end ("1", "2") { + + my $cmd = "cat"; + foreach my $core_name (@core_names) { + + my $file = "$core_name" . "_$end.fastq"; + $cmd .= " $file "; + push (@tmp_files, $file); + } + $cmd .= ">$prefix" . "_$end.fastq"; + &process_cmd($cmd); + } + + ## remove tmp files: + foreach my $file (@tmp_files) { + unlink($file); + } +} + + +exit(0); + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} diff --git a/99.scripts/trinity_utils/util/misc/STAR_align_log_parser.py b/99.scripts/trinity_utils/util/misc/STAR_align_log_parser.py new file mode 100644 index 0000000..e20bb8a --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/STAR_align_log_parser.py @@ -0,0 +1,50 @@ +#!/usr/bin/env python + +import sys, os, re + + + +def main(): + + usage = "\n\n\tusage: {} STAR.Log.final.out [STAR.Log.final.out ...]\n\n".format(sys.argv[0]) + if len(sys.argv) < 2: + exit(usage) + + tokens = ['sample_name'] + + processed_first = False + files = sys.argv[1:] + + for log_filename in files: + + token_to_val = { 'sample_name' : log_filename } + + with open(log_filename) as fh: + for line in fh: + line = line.rstrip() + vals = line.split("|") + if len(vals) != 2: + continue + key = vals[0].strip() + val = vals[1].strip() + if not processed_first: + tokens.append(key) + + token_to_val[key] = val + + if not processed_first: + # print header + print("\t".join(tokens)) + processed_first = True + + + vals = [token_to_val[x] for x in tokens] + print("\t".join(vals)) + + + sys.exit(0) + + + +if __name__=='__main__': + main() diff --git a/99.scripts/trinity_utils/util/misc/TEST_SUPPORT/BFLY_TESTING/graph_out_to_bfly_cmd.pl b/99.scripts/trinity_utils/util/misc/TEST_SUPPORT/BFLY_TESTING/graph_out_to_bfly_cmd.pl new file mode 100644 index 0000000..a650f03 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/TEST_SUPPORT/BFLY_TESTING/graph_out_to_bfly_cmd.pl @@ -0,0 +1,34 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 bfly_graph_out_list.file bfly_jar\n\n"; + +my $bfly_graph_list_file = $ARGV[0] or die $usage; +my $bfly_jar = $ARGV[1] or die $usage; + +main: { + + open (my $fh, $bfly_graph_list_file) or die $!; + while (<$fh>) { + chomp; + my $graph_out_file = $_; + + $graph_out_file =~ s/\.out$//; + + unless (-s "$graph_out_file.reads") { + print STDERR "WARNING - missing $graph_out_file.reads file.... skipping this one.\n"; + next; + } + + my $cmd = "java -Xmx10G -Xms1G -XX:ParallelGCThreads=2 -jar $bfly_jar -N 100000 -L 200 -F 500 -C $graph_out_file --path_reinforcement_distance=75 "; + + print "$cmd\n"; + } + close $fh; + + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/misc/TPM_weighted_gene_length.py b/99.scripts/trinity_utils/util/misc/TPM_weighted_gene_length.py new file mode 100644 index 0000000..9ed3140 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/TPM_weighted_gene_length.py @@ -0,0 +1,157 @@ +#!/usr/bin/env python +# encoding: utf-8 + +from __future__ import (absolute_import, division, + print_function, unicode_literals) +import os, sys, re +import logging +import argparse +import collections + +logging.basicConfig(stream=sys.stderr, level=logging.INFO) +logger = logging.getLogger(__file__) + +def main(): + + parser = argparse.ArgumentParser(description="estimates gene length as isoform lengths weighted by TPM expression values") + + parser.add_argument("--gene_trans_map", dest="gene_trans_map_file", type=str, default="", + required=True, help="gene-to-transcript mapping file, format: gene_id(tab)transcript_id") + + parser.add_argument("--trans_lengths", dest="trans_lengths_file", type=str, required=True, + help="transcript length file, format: trans_id(tab)length") + + parser.add_argument("--TPM_matrix", dest="TPM_matrix_file", type=str, default="", + required=True, help="isoform TPM expression matrix") + + parser.add_argument("--debug", required=False, action="store_true", default=False, help="debug mode") + + + args = parser.parse_args() + + if args.debug: + logger.setLevel(logging.DEBUG) + + + trans_to_gene_id_dict = parse_gene_trans_map(args.gene_trans_map_file) + + trans_lengths_dict = parse_trans_lengths_file(args.trans_lengths_file) + + trans_to_TPM_vals_dict = parse_TPM_matrix(args.TPM_matrix_file) + + weighted_gene_lengths = compute_weighted_gene_lengths(trans_to_gene_id_dict, + trans_lengths_dict, + trans_to_TPM_vals_dict) + + print("#gene_id\tlength") + for gene_id,length in weighted_gene_lengths.items(): + print("\t".join([gene_id,str(length)])) + + + sys.exit(0) + + + +def compute_weighted_gene_lengths(trans_to_gene_id_dict, trans_lengths_dict, trans_to_TPM_vals_dict): + + gene_id_to_trans_list = collections.defaultdict(list) + + gene_id_to_length = {} + + pseudocount = 1 + + for trans_id,gene_id in trans_to_gene_id_dict.items(): + gene_id_to_trans_list[gene_id].append(trans_id) + + for gene_id,trans_list in gene_id_to_trans_list.items(): + + if len(trans_list) == 1: + + gene_id_to_length[gene_id] = trans_lengths_dict[ trans_list[0] ] + else: + + sum_length_x_expr = 0 + sum_expr = 0 + + trans_expr_lengths = [] + + for trans_id in trans_list: + trans_len = trans_lengths_dict[trans_id] + expr_vals = trans_to_TPM_vals_dict[trans_id] + trans_sum_expr = sum(expr_vals) + pseudocount + + trans_expr_lengths.append((trans_len, trans_sum_expr)) + + sum_length_x_expr += trans_sum_expr * trans_len + sum_expr += trans_sum_expr + + + weighted_gene_length = sum_length_x_expr / sum_expr + gene_id_to_length[gene_id] = int(round(weighted_gene_length)) + + logger.debug("Computing weighted length of {0}: {1} => {2}".format(gene_id, + trans_expr_lengths, + weighted_gene_length)) + + return gene_id_to_length + + + +def parse_TPM_matrix(TPM_matrix_file): + + trans_to_TPM_vals_dict = {} + + with open(TPM_matrix_file) as f: + header = next(f) + for line in f: + line = line.rstrip() + vals = line.split("\t") + trans_id = vals[0] + expr_vals_list = vals[1:] + expr_vals_list = [float(x) for x in expr_vals_list] + trans_to_TPM_vals_dict[trans_id] = expr_vals_list + + return trans_to_TPM_vals_dict + + +def parse_trans_lengths_file(trans_lengths_file): + + trans_id_to_length = {} + + with open(trans_lengths_file) as f: + for line in f: + line = line.rstrip() + if line[0] == '#': + continue + + (trans_id, length) = line.split("\t") + + if re.match("^\d+$", length): + trans_id_to_length[trans_id] = int(length) + else: + print("Warning - ignoring line: [{0}] since not parsing length value as number".format(line), file=sys.stderr) + + return trans_id_to_length + + + +def parse_gene_trans_map(gene_trans_map_file): + + trans_to_gene_id = {} + + with open(gene_trans_map_file) as f: + for line in f: + line = line.rstrip() + (gene_id, trans_id) = line.split("\t") + + trans_to_gene_id[trans_id] = gene_id; + + + return trans_to_gene_id + + + +#################### + +if __name__ == "__main__": + main() diff --git a/99.scripts/trinity_utils/util/misc/TophatCufflinksWrapper.pl b/99.scripts/trinity_utils/util/misc/TophatCufflinksWrapper.pl new file mode 100644 index 0000000..d280a10 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/TophatCufflinksWrapper.pl @@ -0,0 +1,199 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use File::Basename; +use FindBin; +use Cwd; + +my $usage = <<__EOUSAGE__; + +############################################################################### +# +# Required: +# +# --target : genome multifasta file +# --seqType :type of reads ('fq' or 'fa') +# +# If paired reads: +# --left :left reads +# --right :right reads +# +# Or, if unpaired reads: +# --single :single reads +# +# --output|-o :name of directory for output +# +# -i :min intron length +# -I :max intron length +# +# +# Optional: +# +# --SS_lib_type :strand-specific library type : {RF, FR, R, F} (RF =~ fr-firststrand) +# --CPU :number of CPUs to use +# +# --GTF :annotations to assist, provided in GTF file format +# +############################################################################### + + +__EOUSAGE__ + + ; + + + +my $help_flag; + +my ($seqType, $left_file, $right_file, $single_file, $SS_lib_type, $output_dir, $CPU, $target_genome, $min_intron_length, $max_intron_length); + +my $GTF_annots; + +&GetOptions ( 'h' => \$help_flag, + + ## general opts + "seqType=s" => \$seqType, + "left=s" => \$left_file, + "right=s" => \$right_file, + "single=s" => \$single_file, + "target=s" => \$target_genome, + + "SS_lib_type=s" => \$SS_lib_type, + "output|o=s" => \$output_dir, + + 'CPU=i' => \$CPU, + 'GTF=s' => \$GTF_annots, + + 'i=i' => \$min_intron_length, + 'I=i' => \$max_intron_length, + + ); + + +if (@ARGV) { + die "Error, do not understand options: @ARGV\n"; +} + +unless ($seqType && $target_genome && ( ($single_file) || ($left_file && $right_file) ) && $output_dir) { + die $usage; +} + +if ($GTF_annots) { + ## add full path + if ($GTF_annots !~ /^\//) { + $GTF_annots = cwd() . "/$GTF_annots"; + } +} + + +main: { + + my $util_dir = "$FindBin::RealBin/.."; + + ############################# + ## align reads using Tophat + ############################# + + my $cmd = "$util_dir/alignReads.pl --seqType $seqType --output $output_dir --aligner tophat2 --target $target_genome "; + if ($single_file) { + $cmd .= " --single $single_file "; + } + else { + $cmd .= " --left $left_file --right $right_file "; + } + if ($SS_lib_type) { + $cmd .= " --SS_lib_type $SS_lib_type "; + } + if ($min_intron_length) { + $cmd .= " -i $min_intron_length "; + } + if ($max_intron_length) { + $cmd .= " -I $max_intron_length "; + } + + + if ($CPU || $GTF_annots) { + $cmd .= " -- "; + + if ($CPU) { + $cmd .= " -p $CPU "; + } + if ($GTF_annots) { + $cmd .= " --GTF $GTF_annots "; + + my $transcriptome_index_dir = "$GTF_annots.tuxedo_indices"; + if (-d $transcriptome_index_dir) { + my $prefix = basename($GTF_annots); + $prefix =~ s/\.[^\.]+$//; + $cmd .= " --transcriptome-index $transcriptome_index_dir/$prefix "; + } + else { + #die "Error, cannot find: $transcriptome_index_dir"; + print STDERR "WARNING: cannot find pre-built index for $GTF_annots; this will slow it down.\n"; + } + + } + } + + &process_cmd($cmd); + + + ####################################### + # assemble transcripts using cufflinks + ####################################### + + my $tophat_alignment_bam = "$output_dir/tophat_out/accepted_hits.bam"; + + $cmd = "cufflinks --no-update-check --upper-quartile-norm -o $output_dir/cuff_out "; + if ($GTF_annots) { + $cmd .= " --GTF-guide $GTF_annots "; + } + + + + + if ($SS_lib_type) { + my %tuxedo_lib_type = (RF => "fr-firststrand", + R => "fr-firststrand", + FR => "fr-secondstrand", + F => "fr-secondstrand", + ); + + my $lib_type = $tuxedo_lib_type{$SS_lib_type} or die "Error, cannot determine tophat lib_type based on SS_lib_type: $SS_lib_type "; + + $cmd .= " --library-type $lib_type "; + } + + if ($CPU) { + $cmd .= " -p $CPU "; + } + + $cmd .= " $tophat_alignment_bam "; + + &process_cmd($cmd); + + exit(0); + +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + + diff --git a/99.scripts/trinity_utils/util/misc/Trinity_genome_aligned_gff3_to_regrouped_genes_gtf.pl b/99.scripts/trinity_utils/util/misc/Trinity_genome_aligned_gff3_to_regrouped_genes_gtf.pl new file mode 100644 index 0000000..9e1063c --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/Trinity_genome_aligned_gff3_to_regrouped_genes_gtf.pl @@ -0,0 +1,348 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use Data::Dumper; +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use SingleLinkageClusterer; +use Overlap_piler; + +my $usage = "usage: $0 gmap.gff3 Trinotate.annot_mappings\n\n"; + +my $gmap_gff3 = $ARGV[0] or die $usage; +my $trinotate_mapping_file = $ARGV[1] or die $usage; + + +my $DEBUG = 0; + +my $GENERIC_GENE_COUNTER = 0; + +my %GENE_NAMES_USED; + +main: { + + my %annotations = &parse_annotation_mappings($trinotate_mapping_file); + + my %contig_to_transcripts = &parse_gmap_gff3($gmap_gff3); + + foreach my $contig (keys %contig_to_transcripts) { + + my @transcript_structs = values %{$contig_to_transcripts{$contig}}; + + # order the exon coords + foreach my $struct (@transcript_structs) { + @{$struct->{coords}} = sort {$a->[0]<=>$b->[0]} @{$struct->{coords}}; + } + + my ($plus_structs_aref, $minus_structs_aref) = &separate_by_strand(@transcript_structs); + + foreach my $struct_list_aref ($plus_structs_aref, $minus_structs_aref) { + + unless (@$struct_list_aref) { next; } # nothing to do + + my @gene_grouped_clusters = &group_genes_by_span_overlaps($struct_list_aref); + + @gene_grouped_clusters = &group_genes_by_exon_overlaps(@gene_grouped_clusters); + + + foreach my $gene_grouped_cluster (@gene_grouped_clusters) { + print STDERR "Contig: $contig, " . Dumper($gene_grouped_cluster) if $DEBUG; + + my ($gene_id_orig, $gene_id_use) = &get_gene_id($gene_grouped_cluster); + + my $gene_name = $annotations{$gene_id_orig} || "$gene_id_orig"; + + foreach my $struct (@$gene_grouped_cluster) { + my $strand = $struct->{strand}; + my $align_name = $struct->{align_name}; + my $transcript_name = $align_name; + $transcript_name =~ s/.mrna\d+//; + + my $trans_annot = $annotations{$transcript_name} || "$transcript_name"; + + my @coords = @{$struct->{coords}}; + foreach my $coordset (@coords) { + print join("\t", $contig, ".", "exon", $coordset->[0], $coordset->[1], ".", $strand, ".", + "transcript_id \"$align_name\"; gene_id \"$gene_id_use\"; transcript_biotype \"processed_transcript\"; gene_name \"$gene_name\"; transcript_name \"$transcript_name\"") . "\n"; + } + print "\n"; # spacer + } + + } + + } + + + } + + + exit(0); + +} + +#### +sub parse_gmap_gff3 { + my ($gmap_gff3_file) = @_; + + ## Example record + ## + # AMEXG_0030000816 AmexG_v3.0.0.fa.gmap gene 4572879 4573094 . - . ID=c104_g1_i1.path1;Name=c104_g1_i1 + # AMEXG_0030000816 AmexG_v3.0.0.fa.gmap mRNA 4572879 4573094 . - . ID=c104_g1_i1.mrna1;Name=c104_g1_i1;Parent=c104_g1_i1.path1;coverage=100.0;identity=100.0;matches=216;mismatches=0;indels=0;unknowns=0 + # AMEXG_0030000816 AmexG_v3.0.0.fa.gmap exon 4572879 4573094 100 - . ID=c104_g1_i1.mrna1.exon1;Name=c104_g1_i1;Parent=c104_g1_i1.mrna1;Target=c104_g1_i1 1 216 + + # AMEXG_0030000816 AmexG_v3.0.0.fa.gmap CDS 4572880 4573092 100 - 0 ID=c104_g1_i1.mrna1.cds1;Name=c104_g1_i1;Parent=c104_g1_i1.mrna1;Target=c104_g1_i1 3 215 + + + + my %contig_to_transcripts; + + ## just capture the exon records. + open(my $fh, $gmap_gff3_file) or die $!; + while(<$fh>) { + if (/^\#/) { next; } + unless (/\w/) { next; } + chomp; + my $line = $_; + my @x = split(/\t/); + my $contig_id = $x[0]; + my $feat_type = $x[2]; + my $lend = $x[3]; + my $rend = $x[4]; + my $strand = $x[6]; + my $info = $x[8]; + + + unless ($feat_type eq "exon") { next; } + + my $align_name; + if ($info =~ /Parent=([^;]+)/) { + $align_name = $1; + } + else{ + die "Error, cannot extract alignment name (Parent) from info: $info of line: $line"; + } + + push (@{$contig_to_transcripts{$contig_id}->{$align_name}->{coords}}, [$lend, $rend]); + + $contig_to_transcripts{$contig_id}->{$align_name}->{strand} = $strand; + $contig_to_transcripts{$contig_id}->{$align_name}->{align_name} = $align_name; + + } + close $fh; + + + return(%contig_to_transcripts); + +} + +#### +sub separate_by_strand { + my (@transcript_structs) = @_; + + my @plus_structs; + my @minus_structs; + + foreach my $struct (@transcript_structs) { + if ($struct->{strand} eq '+') { + push (@plus_structs, $struct); + } + elsif ($struct->{strand} eq '-') { + push (@minus_structs, $struct); + } + else { + die "Error, not sure what strand this corresponds to: " . Dumper($struct); + } + } + + return(\@plus_structs, \@minus_structs); +} + +#### +sub group_genes_by_span_overlaps { + my ($struct_list_aref) = @_; + + print STDERR "\n\n// Grouping by overlaps: " . Dumper($struct_list_aref) . "\n" if $DEBUG; + + my %id_to_struct; + + my $piler = new Overlap_piler(); + + foreach my $struct (@$struct_list_aref) { + my $align_name = $struct->{align_name}; + + $id_to_struct{$align_name} = $struct; + + my @coordsets = @{$struct->{coords}}; + + my @all_coords; + foreach my $coordset (@coordsets) { + my ($lend, $rend) = @$coordset; + push (@all_coords, $lend, $rend); + } + @all_coords = sort {$a<=>$b} @all_coords; + + my $span_lend = shift @all_coords; + my $span_rend = pop @all_coords; + + + $piler->add_coordSet($align_name, $span_lend, $span_rend); + print STDERR "-adding $align_name, $span_lend-$span_rend\n" if $DEBUG; + } + + my @clusters = $piler->build_clusters(); + + print STDERR "Clusters: " . Dumper(\@clusters) . "\n" if $DEBUG; + + my @struct_clusters; + foreach my $cluster (@clusters) { + my @ids = @$cluster; + + my @struct_cluster; + foreach my $id (@ids) { + my $struct = $id_to_struct{$id}; + push (@struct_cluster, $struct); + } + push (@struct_clusters, [@struct_cluster]); + } + + return(@struct_clusters); + +} + +#### +sub get_gene_id { + my ($struct_list_aref) = @_; + + my %genes; + + foreach my $struct (@$struct_list_aref) { + + my $align_name = $struct->{align_name}; + $align_name =~ s/_i\d+.mrna\d+//; + + $genes{$align_name}++; + } + + my @gene_ids = keys %genes; + if (scalar(@gene_ids) == 1 && ! exists $GENE_NAMES_USED{$gene_ids[0]}) { + return($gene_ids[0], $gene_ids[0]); + } + else { + $GENERIC_GENE_COUNTER++; + return($gene_ids[0], "G_$GENERIC_GENE_COUNTER"); + } +} + + +#### +sub parse_annotation_mappings { + my ($trinotate_mapping_file) = @_; + + my %annots; + + open(my $fh, $trinotate_mapping_file) or die $!; + while(<$fh>) { + chomp; + my ($feature_id, $annot) = split(/\t/); + $annots{$feature_id} = $annot; + } + close $fh; + + return(%annots); +} + +#### +sub group_genes_by_exon_overlaps { + my (@groups) = @_; + + my @groups_ret = (); + + foreach my $group (@groups) { + if (scalar(@$group) == 1) { + push (@groups_ret, $group); + } + else { + my @refined_groups = &partition_by_exon_overlap(@$group); + push (@groups_ret, @refined_groups); + } + } + + return(@groups_ret); +} + +#### +sub partition_by_exon_overlap { + my @structs = @_; + + my %id_to_struct; + foreach my $struct (@structs) { + $id_to_struct{$struct->{align_name}} = $struct; + } + + my @pairs; + + for(my $i = 0; $i < $#structs; $i++) { + my $struct_i = $structs[$i]; + + for (my $j = $i + 1; $j <= $#structs; $j++) { + my $struct_j = $structs[$j]; + + if (&exon_overlap($struct_i, $struct_j)) { + + push (@pairs, [$struct_i->{align_name}, $struct_j->{align_name}]); + } + } + } + + + my @clusters; + if (scalar(@pairs) > 1) { + @clusters = &SingleLinkageClusterer::build_clusters(@pairs); + } + elsif (scalar @clusters == 1) { + @clusters = @pairs; + } + + my %seen; + + my @clusters_ret; + foreach my $cluster (@clusters) { + my @cluster_ret; + foreach my $ele (@$cluster) { + my $struct = $id_to_struct{$ele} or die "Error, no struct for $ele"; + push (@cluster_ret, $struct); + $seen{$ele} = 1; + } + push (@clusters_ret, \@cluster_ret); + } + + foreach my $struct (@structs) { + if (! exists $seen{$struct->{align_name}}) { + push (@clusters_ret, [$struct]); + } + } + + return(@clusters_ret); +} + +#### +sub exon_overlap { + my ($struct_A, $struct_B) = @_; + + print "A: " . Dumper($struct_A) . "\nB: " . Dumper($struct_B) . "\n" if $DEBUG; + + my @exon_coords_A = @{$struct_A->{coords}}; + my @exon_coords_B = @{$struct_B->{coords}}; + + foreach my $exon_A (@exon_coords_A) { + foreach my $exon_B (@exon_coords_B) { + + if ($exon_A->[0] < $exon_B->[1] && $exon_A->[1] > $exon_B->[0]) { + return(1); # overlap detected + } + } + } + + return(0); # no overlap +} + diff --git a/99.scripts/trinity_utils/util/misc/Trinity_node_seq_extractor.pl b/99.scripts/trinity_utils/util/misc/Trinity_node_seq_extractor.pl new file mode 100644 index 0000000..152cca9 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/Trinity_node_seq_extractor.pl @@ -0,0 +1,79 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + +my $usage = "usage: $0 fasta\n"; + +my $fasta_file = $ARGV[0] or die $usage; + +my $fasta_reader = new Fasta_reader($fasta_file); + + +my %node_id_to_seq; + +while (my $seq_obj = $fasta_reader->next()) { + + my $header = $seq_obj->get_header(); + my $seq = $seq_obj->get_sequence(); + my $accession = $seq_obj->get_accession(); + + my $gene_id = $accession; + $gene_id =~ s/_i\d+$//; + + + my $path_info; + + if ($header =~ /path=\[([^\]]+)\]/) { + $path_info = $1; + } + else { + die "Error, cannot extract path info from $header"; + } + + #print "path: $path_info\n"; + + my @nodes = split(/\s+/, $path_info); + for my $node (@nodes) { + my ($node_id, $seq_range) = split(/:/, $node); + + $node_id = join(":", $gene_id, $node_id); + + my ($lend, $rend) = split(/-/, $seq_range); + my $node_seq = substr($seq, $lend, $rend - $lend + 1); + + #print "$node_id\t$node_seq\n"; + + if (exists $node_id_to_seq{$node_id}) { + if ($node_id_to_seq{$node_id} ne $node_seq) { + + my ($longer_seq, $shorter_seq) = reverse sort {length($a)<=>length($b)} ($node_seq, $node_id_to_seq{$node_id}); + if (index($longer_seq, $shorter_seq) > 0) { + # keep the longer one. + $node_id_to_seq{$node_id} = $longer_seq; + } + else { + print STDERR "-warning, difference in node seqs: $node_id\t$node_id_to_seq{$node_id}\tvs\t$node_seq\n"; + } + } + } + else { + $node_id_to_seq{$node_id} = $node_seq; + } + + } +} + + +for my $node_id (sort keys %node_id_to_seq) { + my $node_seq = $node_id_to_seq{$node_id}; + print "$node_id\t" . length($node_seq) . "\t$node_seq\n"; +} + + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/acc_list_to_fasta_entries.pl b/99.scripts/trinity_utils/util/misc/acc_list_to_fasta_entries.pl new file mode 100644 index 0000000..843bcb0 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/acc_list_to_fasta_entries.pl @@ -0,0 +1,226 @@ +#!/usr/bin/env perl + +# works for trinity transcripts or trinity genes, Sept 28 2016 bhaas + + +# lightweight fasta reader capabilities: +package Fasta_reader; + +use strict; + +sub new { + my ($packagename, $fastaFile) = @_; + + ## note: fastaFile can be a filename or an IO::Handle + + + my $self = { fastaFile => undef,, + fileHandle => undef }; + + bless ($self, $packagename); + + ## create filehandle + my $filehandle = undef; + + if (ref $fastaFile eq 'IO::Handle') { + $filehandle = $fastaFile; + } + else { + + open ($filehandle, $fastaFile) or die "Error: Couldn't open $fastaFile\n"; + $self->{fastaFile} = $fastaFile; + } + + $self->{fileHandle} = $filehandle; + + return ($self); +} + + + +#### next() fetches next Sequence object. +sub next { + my $self = shift; + my $orig_record_sep = $/; + $/="\n>"; + my $filehandle = $self->{fileHandle}; + my $next_text_input = <$filehandle>; + + if (defined($next_text_input) && $next_text_input !~ /\w/) { + ## must have been some whitespace at start of fasta file, before first entry. + ## try again: + $next_text_input = <$filehandle>; + } + + my $seqobj = undef; + + if ($next_text_input) { + $next_text_input =~ s/^>|>$//g; #remove trailing > char. + $next_text_input =~ tr/\t\n\000-\037\177-\377/\t\n/d; #remove cntrl chars + my ($header, @seqlines) = split (/\n/, $next_text_input); + my $sequence = join ("", @seqlines); + $sequence =~ s/\s//g; + + $seqobj = Sequence->new($header, $sequence); + } + + $/ = $orig_record_sep; #reset the record separator to original setting. + + return ($seqobj); #returns null if not instantiated. +} + + +#### finish() closes the open filehandle to the query database. +sub finish { + my $self = shift; + my $filehandle = $self->{fileHandle}; + close $filehandle; + $self->{fileHandle} = undef; +} + +#### +sub retrieve_all_seqs_hash { + my $self = shift; + + my %acc_to_seq; + + while (my $seq_obj = $self->next()) { + my $acc = $seq_obj->get_accession(); + my $sequence = $seq_obj->get_sequence(); + + $acc_to_seq{$acc} = $sequence; + } + + return(%acc_to_seq); +} + + + +############################################## +package Sequence; +use strict; + +sub new { + my ($packagename, $header, $sequence) = @_; + + ## extract an accession from the header: + my ($acc, $rest) = split (/\s+/, $header, 2); + + my $self = { accession => $acc, + header => $header, + sequence => $sequence, + filename => undef }; + bless ($self, $packagename); + return ($self); +} + +#### +sub get_accession { + my $self = shift; + return ($self->{accession}); +} + +#### +sub get_header { + my $self = shift; + return ($self->{header}); +} + +#### +sub get_sequence { + my $self = shift; + return ($self->{sequence}); +} + +#### +sub get_FASTA_format { + my $self = shift; + my $header = $self->get_header(); + my $sequence = $self->get_sequence(); + $sequence =~ s/(\S{60})/$1\n/g; + my $fasta_entry = ">$header\n$sequence\n"; + return ($fasta_entry); +} + + +#### +sub write_fasta_file { + my $self = shift; + my $filename = shift; + + my ($accession, $header, $sequence) = ($self->{accession}, $self->{header}, $self->{sequence}); + + my $fasta_entry = $self->get_FASTA_format(); + + my $tempfile; + if ($filename) { + $tempfile = $filename; + } else { + my $acc = $accession; + $acc =~ s/\W/_/g; + $tempfile = "$acc.fasta"; + } + + open (TMP, ">$tempfile") or die "ERROR! Couldn't write a temporary file in current directory.\n"; + print TMP $fasta_entry; + close TMP; + return ($tempfile); +} + +package main; + +my $usage = "usage: $0 acc.list.txt file.fasta\n\n"; + +my $acc_list = $ARGV[0] or die $usage; +my $pep = $ARGV[1] or die $usage; + +main: { + my $fasta_reader = new Fasta_reader($pep); + + my $acc_text = `cat $acc_list`; + + my %accs; + + while ($acc_text =~ /(\S+)/g) { + my $acc = $1; + + $accs{$acc} = 1; + } + + my %seen; + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + + my $gene_id = $acc; + $gene_id =~ s/_i\d+$//; + + if ($accs{$acc} || $accs{$gene_id}) { + print $seq_obj->get_FASTA_format(); + $seen{$acc} = 1 if $accs{$acc}; + $seen{$gene_id} = 1 if $accs{$gene_id}; + + } + } + + # remove seen entries + foreach my $seen_acc (keys %seen) { + delete $accs{$seen_acc} if exists $accs{$seen_acc}; + } + + + if (%accs) { + print STDERR "Error, could not locate entries for: " . join(", ", keys %accs) . "\n"; + exit(1); + } + + exit(0); +} + + + + + + + + diff --git a/99.scripts/trinity_utils/util/misc/alexie_analyze_blast.pl b/99.scripts/trinity_utils/util/misc/alexie_analyze_blast.pl new file mode 100644 index 0000000..f0cdcbb --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/alexie_analyze_blast.pl @@ -0,0 +1,870 @@ +#!/usr/bin/env perl + +use warnings; + +#CHANGELOG +#05Jun09 - Added option to ignore X/Ns when estimating size of Query +# (to allow blasting specific strains of an assembly only after convert_project and replacing @s with Xs) +# Provide the initial fastafile used to blast or +#11Jun09 - Added option to extract hits and misses ('CDS' and 'UTR') +# +# +# +# +our $VERSION = '1.5'; + +#### +### TODO: add dbname on the top entry of the hashes so that we can have multiple dbs in same file? +# TODO: implement DBM to store hash http://www.perl.com/pub/a/2006/02/16/mldbm.html + +=head1 NAME + +analyze_assembly_blast.pl - Provides some simple descriptive stats in a SearchIO BLAST report. +Useful for example to investigate the coverage of an experiment-derived FASTA file A versus a reference FASTA file B. + +=head1 VERSION + + Version 1.4 + +=head1 USAGE + + analyze_assembly_blast.pl [-cs -ce -l ] + + -cs|--cutoff-score:i Specify cutoff score for report. Defaults to 80 + -ce|--cutoff-evalue:s Specify cutoff evalue for report. Defaults to 1e-5 + -f|--format:s Specify format (currently only BLAST supported) + -d Debug output. -dd is more detailed (only small reports: high memory usage) + -l|--limit:i Limit processing to these many top hits for each query. Will also limit the hash output (so cannot reuse a hash for a larger limit) + -uh|--use_hash If for some reason you want to redo the search (e.g. change cutoff, reduce limit etc), + this switch will speed things up significantly. No need to provide the original BLAST file. Can be used for more stringent -cs/-ce than the one used to build the hash, but clearly, cannot use a less stringent -cs/-ce + -nt|--notimer Specify to switch off timer. + --lines:i Specify number of lines in hash (to speed up timer stats) + -x|maskxn Don't include Ns or Xs when calculating query sequence length. Give FASTA used to BLAST or a number to set size of query. + -extract:s If you give the FASTA used as input to BLAST, I can give you the regions matching and not matching as FASTA. Cannot be used with -uh + +=head1 AUTHORS + + Alexie Papanicolaou 1 2 + with help from Paul Wilkinson 2 + + 1 Max Planck Institute for Chemical Ecology, Germany + 2 Centre for Ecology and Conservation, University of Exeter, UK + alexie@butterflybase.org + +=head1 DISCLAIMER & LICENSE + +This software is released under the GNU General Public License version 3 (GPLv3). +It is provided "as is" without warranty of any kind. +You can find the terms and conditions at http://www.opensource.org/licenses/gpl-3.0.html. +Please note that incorporating the whole software or parts of its code in proprietary software +is prohibited under the current license. + +=head1 BUGS & LIMITATIONS + +Unfortunately tiles hsps only works with normal blast report, not tabulated output... +also, tblastx is a bit tricky... not sure if tiled works properly there (blastn is fine) + +Verification from Jason S and Chris Field. BioPerl tiling is crap, best to use wu-blast apparently +with the -links option. + +=cut + + +use strict; +use warnings; +use Pod::Usage; +use Bio::SeqIO; +use Bio::SearchIO; +use Bio::Search::SearchUtils; +use Bio::Index::Fasta; +use Statistics::Descriptive; +#use Time::Progress; +#my $timer = new Time::Progress; +use Getopt::Long; +$| = 1; + +# Declare options +my ( + $debug, $debug2, + $limit, $use_hash, $multiple_blasts, + $debugfile, $notimer, $no_check +); +my $store_hash = 1; # forced because queries depend on it. +my $cut_score = 80; +my $cut_evalue = '1e-5'; +my $report_style = "blast"; +my $hash_lines; + +# other global variables +my ( $query_tlength, $query_tlength_with_hits,$idfile,%ids,$maskxn,$extract,$extract_inx,%extr_hash,$single_copy ); +GetOptions( + 'cs|cutoff-score:i' => \$cut_score, + 'ce|cutoff-evalue:s' => \$cut_evalue, + 'f|format:s' => \$report_style, + 'd' => \$debug, + 'dd' => \$debug2, + 'l|limit:i' => \$limit, # process this top hits + 'id|idfile:s' => \$idfile, + 'uh|use_hash' => \$use_hash, + 'nt|notimer' => \$notimer, + 'nc|nocheck' => \$no_check, + 'lines:i' => \$hash_lines, + 'x|maskxn:s' => \$maskxn, + 'extract:s' => \$extract, # file to extract fasta from into CDS + 'single' => \$single_copy +); +if ( !@ARGV ) { + warn("\nUnless you give me at least one BLAST file, there is nothing for me to do!\n"); + pod2usage; + +} +if ($use_hash) { undef($store_hash); } +my @blastfiles = @ARGV; +if ($debug2) { $debug = 1; } + +if ($extract){ + unless (-s $extract){die ("Could not find $extract");} + unless (-f "$extract.index"){ + print "Indexing $extract...\n"; + my $inx = Bio::Index::Fasta->new(-filename => "$extract.index",-write_flag => 1); + $inx->make_index($extract); + } + $extract_inx = Bio::Index::Fasta->new(-filename => "$extract.index") || die ("Could not get index for $extract\n"); +} + + +if ($idfile && -s $idfile) { + my $pattern; + $pattern='/^\s*(\S+)\s+/'; + my @test_lines=`head $idfile`; + foreach my $test (@test_lines){if ($test=~/^>/){$pattern="Bio::SeqIO";}} + print "Building hash from $idfile with $pattern\n"; + + if ($pattern eq "Bio::SeqIO"){ + my $id_obj=new Bio::SeqIO( -file => $idfile,-format => "fasta" ); + while ( my $seq = $id_obj->next_seq() ) { + $ids{$seq->id()}=1; + } + } + else{ + open( IN, $idfile ) || die(); + my $flag = 0; + while ( my $line = ) { + if ( $line =~ $pattern ) { + $ids{$1} = 1; + if ( $flag == 0 ) { + print "Hash presence of $idfile verified\n"; + $flag = 1; + } + } + } + close(IN); + } +} + +foreach my $blastfile (@blastfiles) { + if ($use_hash && $blastfile=~/^(.+)\.hash$/){$blastfile=$1;} + unless ( -s $blastfile || -s "$blastfile.hash" ) { + die("Can't find $blastfile\n"); + } + print "Processing $blastfile...\n"; + my $logfile=$blastfile.".analysis.".$cut_score.".".$cut_evalue; + #reset global varialbes + $query_tlength = int(0); + $query_tlength_with_hits = int(0); + if ($use_hash) { + unless ( -e "$blastfile.hash" ) { + die("You requested to use a previous hash but I can't find $blastfile.hash\n" );} + } + if ( $debug || $debug2 ) { + use Data::Dumper; + $debugfile = $blastfile . ".debug"; + open( DEBUG, ">$debugfile" ); + } + open( LOG, ">$logfile" ); + if ($store_hash) { + if ( -s "$blastfile.hash" ) { + warn "$blastfile.hash already exists. Are you sure you want to rebuild the HASH-table?\nWait 1sec if yes; Ctl-C otherwise\n"; + sleep(1); + } + open( HASH, ">$blastfile.hash" ); + print HASH "TYPE\tSEQ NAME\tSEQ LENGTH\tGLOBAL HSP No\tIDENTICAL\tCONSERVED\tSTART\tEND\tBITSCORE\tEVALUE\tLENGTH PROP\tLOCAL HSP No\tLOCAL HIT No\tDIRECTION\n"; + } + &process_blast($blastfile); + if ($store_hash) { close(HASH); } + + if ($extract){ + print "Preparing CDS and UTR files...\n"; + my $fasta_cds = new Bio::SeqIO( -file => ">$extract.putative_cds.$blastfile", -format => "fasta" ); + $fasta_cds->width(15000); + my $fasta_utr = new Bio::SeqIO( -file => ">$extract.putative_utr.$blastfile", -format => "fasta" ); + $fasta_utr->width(15000); + foreach my $query_name (keys %extr_hash){ + my $gene_obj = $extract_inx->fetch($query_name); + die "Cannot find sequence $query_name in $extract\n" if !$gene_obj; + die "Sequence seems to be a protein (or have IUPAC codes), cannot create CDS/UTR files!\n" if ($gene_obj->alphabet() ne 'dna') ; + if (!$gene_obj){next;} + my $gene_length=$gene_obj->length(); + my $id=$gene_obj->id(); + my $start=$extr_hash{$query_name}{"start"}; + my $end=$extr_hash{$query_name}{"end"}; + my $direction = $extr_hash{$query_name}{"direction"}; + #$gene_obj = $gene_obj->revcom() if $direction eq 'R'; + if (!$start){$start=$gene_length+1;} + if (!$end){$end=0;} + #print "Processing id $id length $gene_length. CDS start $start end $end \n"; + my $cds=new Bio::Seq(); + my $utr5=new Bio::Seq(); + my $utr3=new Bio::Seq(); + my $cds_seq = $direction eq 'R' ? &revcomp($gene_obj->subseq($start,$end)) : $gene_obj->subseq($start,$end); + $cds->seq($cds_seq); + my $new_id=$id; + my $desc = "CDS ".$start."_".$end; + $desc.=' R' if $direction eq 'R'; + $cds->desc($desc); + $cds->id($new_id); + $fasta_cds->write_seq($cds); + if ($start>2){ + my $new_start=1; + my $new_end=$start-1; + my $seq = $direction eq 'R' ? &revcomp($gene_obj->subseq($new_start,$new_end)) : $gene_obj->subseq($new_start,$new_end); + $utr5->seq($gene_obj->subseq($new_start,$new_end)); + my $new_id=$id; + my $desc = "5UTR ".$new_start."_".$new_end.' F' if $direction eq 'F'; + $desc = "3UTR ".$new_start."_".$new_end.' R' if $direction eq 'R'; + $utr5->desc($desc); + $utr5->id($new_id); + $fasta_utr->write_seq($utr5); + $fasta_utr->write_seq($utr5); + } + if ($end<$gene_length){ + my $new_start=$end+1; + my $new_end=$gene_length; + $utr3->seq($gene_obj->subseq($new_start,$new_end)); + my $new_id=$id; + my $desc = "3UTR ".$new_start."_".$new_end.' F' if $direction eq 'F'; + $desc = "5UTR ".$new_start."_".$new_end.' R' if $direction eq 'R'; + $utr5->desc($desc); + $utr3->id($new_id); + $fasta_utr->write_seq($utr3); + } + } + } + + + + #print $print_statement; + #print LOG $print_statement; + close(LOG); + close(DEBUG); + print "All Done! See $logfile\n#==============================#\n\n"; + +} +################################################################## +# SUBROUTINES +################################################################## +sub process_blast($) { + my $blastfile = shift; + print "Parsing BLAST report $blastfile\n"; + unless ($use_hash) { + my $total_queries = `grep -c Query= $blastfile`; + chomp($total_queries); + #$timer->attr( min => 0, max => $total_queries ); + #$timer->restart; + + # verify blast has run to completion + unless ($no_check) { + if ( $report_style eq 'blast' || $report_style eq 'BLAST' ) { + my $result; + my @lines = `tail -n 20 $blastfile`; + foreach my $ln (@lines){ + $result = 1 if $ln=~/^Matrix/; + } + if ( !$result ) { + print("Sorry, but it seems your BLAST report is incomplete\n"); + return; + } + } + } + } + my ( $hash_ref_db_elements, $hash_ref_queries, $db_name, $db_length, + $db_entries, $query_total ); + my $annotation_redundancy=int(0); +# to reduce memory we can split the read hash in two rounds and build each hash independently then emptying it. +# but we only want to read the blast file once so the read blast does no longer build the hash, but it does print it +# to be read later + if ($use_hash) { + print "Reading database hash\n"; + ( + $hash_ref_db_elements, $db_name, $db_length, $db_entries, + $query_total + ) = &read_hash( $blastfile, "database" ); + } else { + print "Building new database hash\n"; + ( + $hash_ref_db_elements, $db_name, $db_length, $db_entries, + $query_total + ) = &build_hash($blastfile); + } + + + ## Now with hashes built continue to process things + my ( $db_ided, $query_ided ); + + # prepare arrays for the Stats + my $db_with_hit=int(0); + my ($conserved_array_ref, $identical_array_ref, $unique_matches_ref,$all_matches_ref,$length_proportions_ref); + if ($hash_ref_db_elements) { + print "Processing database data\n"; + my $print_statement; + # memory explodes here. + ( $conserved_array_ref, $identical_array_ref, $unique_matches_ref, + $all_matches_ref, $length_proportions_ref, $db_with_hit + ) = parse_blasthash( $hash_ref_db_elements, "database",$blastfile ); + $db_ided = $db_with_hit; + my $db_with_hit_ratio = sprintf( "%d %%", $db_with_hit / $db_entries * 100 ); + undef(%$hash_ref_db_elements); # is this correct way to empty memory? + undef($hash_ref_db_elements); + $print_statement .="\nYou searched versus\t$db_name:\nTotal entries:\t$db_entries\nDatabase positions:\t$db_length positions\nIdentified:\t$db_with_hit\t($db_with_hit_ratio).\n"; + print "."; + foreach my $prop (sort keys %{$length_proportions_ref}){ + $print_statement .="Database entries identified with total length at least \t".$prop."%\t".$length_proportions_ref->{$prop}."\n"; + } + print "."; + my @unique_matches = @$unique_matches_ref; + my ( $unique_sum, $unique_mean, $unique_sd, $unique_median ) = &prepare_stats( \@unique_matches, "unique" ); + $unique_mean = sprintf( "%d", $unique_mean ); + my $prop_db_identified = + sprintf( "%d %%", $unique_sum / $db_length * 100 ); + $print_statement .="Non-overlapping positions identified:\t$unique_sum ($prop_db_identified), with mean length $unique_mean (SD=$unique_sd) and median $unique_median\n"; + undef(@unique_matches); + print "."; + my @all_matches = @$all_matches_ref; + my ( $all_sum, $all_mean, $all_sd, $all_median ) = &prepare_stats( \@all_matches, "all" ); + $all_mean = sprintf( "%d", $all_mean ); + my $unique_ratio = sprintf( "%.6f", $all_sum / $unique_sum ); + $annotation_redundancy+=$unique_ratio; + $print_statement .="Overlapping positions identified:\t$all_sum, with mean length $all_mean (SD=$all_sd) and median $all_median.\nOverlapping/non-overlapping ratio:\t$unique_ratio\n"; + undef(@all_matches); + print "."; + my @conserved_array = @$conserved_array_ref; + my ( $conserved_sum, $conserved_mean, $conserved_sd, $conserved_median ) = &prepare_stats( \@conserved_array, "conserved" ); + $conserved_mean = sprintf( "%.2f %%", $conserved_mean ); + $print_statement .="Mean of conserved positions:\t$conserved_mean (SD=$conserved_sd), and median $conserved_median\n"; + undef(@conserved_array); + print "."; + my @identical_array = @$identical_array_ref; + my ( $identical_sum, $identical_mean, $identical_sd, $identical_median ) + = &prepare_stats( \@identical_array, "identical" ); + $identical_mean = sprintf( "%.2f %%", $identical_mean ); + $print_statement .="Mean of identical positions:\t$identical_mean (SD=$identical_sd), and median $identical_median\n"; + undef(@identical_array); + print ". Done\n"; + if ($debug) { &create_debug_log( $hash_ref_db_elements, "database" ); } + print LOG $print_statement; + } + ( $hash_ref_queries, $db_name, $db_length, $db_entries, $query_total ) = &read_hash( $blastfile, "queries" ); + + #QUERY data + # update Query length if X masking requested + if ($maskxn){ + $query_tlength=int(0); + if (int($maskxn)){ + print "Query length provided by user as $maskxn.\n"; + $query_tlength=$maskxn; + } + else { + my $fasta_obj=new Bio::SeqIO( -file => $maskxn,-format => "fasta" ) || die("$maskxn is not a fasta file\n"); + while ( my $seq = $fasta_obj->next_seq() ) { + my $sequence=$seq->seq(); + $sequence=~s/X+//ig; + $query_tlength+=length($sequence); + } + print "Query length estimated from inputfile as query_tlength. No X/Ns\n"; + } + if (!$query_tlength || $query_tlength==0){ + die "Query length is 0. This should not have happened. Did you give the -maskx argument properly?\n"; + } + } +# \@conserved_array, \@identical_array, \@unique_matches,\@all_matches, \%length_proportions, $element_number + if ($hash_ref_queries) { + print "Processing query data\n"; + my $print_statement; + my ( $unique_matches_ref, $all_matches_ref, $length_proportions_ref ,$queries_with_hit) = parse_blasthash( $hash_ref_queries, "queries",$blastfile ); + $query_ided = $queries_with_hit; + my $queries_with_hit_ratio = sprintf( "%d %%", $queries_with_hit / $query_total * 100 ); + undef(%$hash_ref_queries); + undef($hash_ref_queries); + $print_statement .="\nFrom the query perspective:\nTotal queries: $query_total. Queries with hit:\t$queries_with_hit ($queries_with_hit_ratio)\n"; + print '.'; + foreach my $prop (sort keys %{$length_proportions_ref}){ + $print_statement .="Query sequences that are covered by reference at least \t".$prop."%\t".$length_proportions_ref->{$prop}."\n"; + } + print '.'; + #my @conserved_array=@$conserved_array_ref; + #my @identical_array=@$identical_array_ref; + my @unique_matches = @$unique_matches_ref; + my @all_matches = @$all_matches_ref; + my ( $unique_sum, $unique_mean, $unique_sd, $unique_median ) = &prepare_stats( \@unique_matches, "unique" ); + $unique_mean = sprintf( "%d", $unique_mean ); + my $prop_query_identified = sprintf( "%d %%", $unique_sum / $query_tlength * 100 ); + $print_statement .= "Total length:\t$query_tlength positions.\nNon-overlapping positions identified:\t$unique_sum ($prop_query_identified), with mean length $unique_mean (SD=$unique_sd) and median $unique_median\n"; + undef(@unique_matches); + print "."; + my ( $all_sum, $all_mean, $all_sd, $all_median ) = &prepare_stats( \@all_matches, "all" ); + $all_mean = sprintf( "%d", $all_mean ); + my $unique_ratio = sprintf( "%.6f ", $all_sum / $unique_sum ); + $annotation_redundancy+=$unique_ratio; + $print_statement .="Overlapping positions identified:\t$all_sum, with mean length $all_mean (SD=$all_sd) and median $all_median.\nOverlapping/non-overlapping ratio:\t$unique_ratio\n"; + undef(@all_matches); + print ". Done\n"; + $print_statement .="\nTotal annotation redundancy:\t$annotation_redundancy\n"; + if ($debug) { &create_debug_log( $hash_ref_queries, "Queries" ); } + print LOG $print_statement; + } + return (0); # success! +} + +sub prepare_stats($$) { + my $array_ref = shift; + my $array_type = shift; + my @array = @$array_ref; + my $print_statement; + if ($debug) { print DEBUG "$array_type\n"; print DEBUG Dumper @array; } + my $stat = Statistics::Descriptive::Full->new(); + $stat->add_data(@array); + my $mean = $stat->mean(); + if ($mean) { $mean = sprintf( "%.4f", $mean ); } + else { $mean = 0; } + my $sd = $stat->standard_deviation(); + if ($sd) { $sd = sprintf( "%.2f", $sd ); } + else { $sd = 0; } + my $median = $stat->median(); + if ( !$median ) { $median = 0; } + my $sum = $stat->sum(); + return ( $sum, $mean, $sd, $median ); +} + +sub parse_blasthash ($$$) { + my $hash_ref = shift; # this either the db elements or the query elements + my $hash_type = shift; # database or queries + my $blastfile = shift; + my @conserved_array; # for database only + my @identical_array; # for database only + my @unique_matches; + my @all_matches; + my %length_proportions; + my $element_number = int(0); # hit or queries present in hash + open (FULL_LENGTH,">$blastfile.$hash_type.full"); + foreach my $element ( keys %$hash_ref ) { + + # this is not really necessary, unless we want to build a graph later on (or debug) + if ($debug2) { + my $length = $hash_ref->{$element}{"length"}; + for ( my $i = 1 ; $i <= $length ; $i++ ) { + $hash_ref->{$element}{"pos"}{$i} = int(0); + } + } + # for each database element, we have top aln_prop + #unless ( $hash_type eq "queries" ) { + if ($hash_ref->{$element}{"aln_prop"}){ + print FULL_LENGTH $element."\t".$hash_ref->{$element}{'align_name'}."\t".$hash_ref->{$element}{"aln_prop"}."\n" if $hash_ref->{$element}{"aln_prop"}>=0.80; + $length_proportions{25}++ if $hash_ref->{$element}{"aln_prop"}>=0.25; + $length_proportions{50}++ if $hash_ref->{$element}{"aln_prop"}>=0.50; + $length_proportions{60}++ if $hash_ref->{$element}{"aln_prop"}>=0.60; + $length_proportions{70}++ if $hash_ref->{$element}{"aln_prop"}>=0.70; + $length_proportions{75}++ if $hash_ref->{$element}{"aln_prop"}>=0.75; + $length_proportions{80}++ if $hash_ref->{$element}{"aln_prop"}>=0.80; + $length_proportions{90}++ if $hash_ref->{$element}{"aln_prop"}>=0.90; + $length_proportions{95}++ if $hash_ref->{$element}{"aln_prop"}>=0.95; + } + #} + # memory explode + # go to each hsp and get data out + + foreach my $hsp ( keys %{ $hash_ref->{$element}{"hsp"} } ) { + # first do the positions of element. + my $element_start_position = $hash_ref->{$element}{"hsp"}{$hsp}{"start"}; + my $element_end_position = $hash_ref->{$element}{"hsp"}{$hsp}{"end"}; + for ( my $i = $element_start_position ; + $i <= $element_end_position ; + $i++ ){ + my $old_size = $hash_ref->{$element}{"pos"}{$i} if ($debug2); + $hash_ref->{$element}{"pos"}{$i}++; + if ($debug2) { + my $new_size = $hash_ref->{$element}{"pos"}{$i}; + print DEBUG "\nFound one! $element ($element_start_position,$element_end_position): $i was $old_size and is $new_size\n"; + } + } + + # now build up conserved/identical arrays + unless ( $hash_type eq "queries" ) { + push( @conserved_array, + $hash_ref->{$element}{"hsp"}{$hsp}{"conserved"} ); + push( @identical_array, + $hash_ref->{$element}{"hsp"}{$hsp}{"identical"} ); + } + } + +# we finished cycling through the HSPs of this element so if we want to do stats to the positions of this element +# only, then this is the place to do it. At the moment we don't; so move on. + } + close (FULL_LENGTH); + print "Hash parsed. Populating arrays...\n"; + +# Positions on subjects have been stored so we can now accumulate the results and populate the arrays +# Now with positions, we could see how the exact coverage per base is too... for future implementation; + foreach my $element ( keys %$hash_ref ) { + $element_number++; + foreach my $position ( keys %{ $hash_ref->{$element}{"pos"} } ) { + $hash_ref->{$element}{"aggregate_total"} += + $hash_ref->{$element}{"pos"}{$position}; + $hash_ref->{$element}{"unique_total"}++; + } + + # avoid undefs going in! + my $match = $hash_ref->{$element}{"aggregate_total"}; + my $unique = $hash_ref->{$element}{"unique_total"}; + if ($match) { push( @all_matches, $match ); } + if ($unique) { push( @unique_matches, $unique ); } + } + if ( $hash_type eq "queries" ) { + return ( \@unique_matches, \@all_matches, \%length_proportions,$element_number ); + } elsif ( $hash_type eq "database" ) { + return ( + \@conserved_array, \@identical_array, \@unique_matches, + \@all_matches, \%length_proportions, $element_number + ); + } +} + +# use stored hash +sub read_hash($$) { + my $blastfile = shift; + my $hash_type = shift; + warn "Reading hash $hash_type\n"; + my ( %hash, $db_name, $db_length, $db_entries, $query_total ); + print "Stored HASH found, parsing...\n"; + open( HASH, "$blastfile.hash" ); + my (%hit_counting, %query_counting); + my $total_hit_counter=int(0); + my $hsp_counter=int(0); + my $query_counter=int(0); + #bookmark + # timer + my $line_num=int(0); + if ($hash_lines) { $line_num = $hash_lines; } + elsif (-s "$blastfile.hash") { + $line_num = `wc $blastfile.hash`; + chomp($line_num); + $line_num =~ s/^\s*(\d+)\s.+$/$1/; + } + print "HASH has $line_num lines...\n"; + my $line_counter; + #$timer->attr( min => 0, max => $line_num ); + #$timer->restart; + LINE: while ( my $line = ) { + $line_counter++; + #if ( !$notimer && $line_counter =~ /0000$/ ) { + # print $timer->report( "eta: %E min, %40b %p\r", $line_counter ); + #} + if ( $line =~ /^\#/ ) { + if ( $line =~ /^\#Total.+\:(\d+)/ ) { + $query_tlength_with_hits = $1; + } elsif ( $line =~ /^\#Overall.+\:(\d+)/ ) { + $query_tlength = $1; + } elsif ( $line =~ /^\#Queries total:(\d+)/ ) { + $query_total = $1; + } elsif ( $line =~ /^\#DB/ ) { + if ( $line =~ /name:(\S+)/ ) { $db_name = $1; } + elsif ( $line =~ /length\:(\d+)/ ) { $db_length = $1; } + elsif ( $line =~ /entries\:(\d+)/ ) { $db_entries = $1; } + } else { + next; + } + } + chomp($line); + my @data = split( "\t", $line ); + if ( @data && $hash_type eq "database" && $data[0] eq "HIT" ) { + my $db_element_name = $data[1]; + my $db_element_length = $data[2]; + $hsp_counter = $data[3]; + my $frac_id_db = $data[4]; + my $frac_cons_db = $data[5]; + my $hstart = $data[6]; + my $hend = $data[7]; + my $score = $data[8]; + my $eval = $data[9]; + my $ref_aln_prop = $data[10]; + my $local_hsp_counter = $data[11]; + my $local_hit_counter = $data[12] if $data[12]; + my $direction = $data[13]; + my $query_name = $data[14]; + next if ($limit && $local_hit_counter && $limit < $local_hit_counter); + # a) to prevent low scoring HSPs to be accepted - repeats; b) to allow reparsing of hash. Next line. + if ( $score < $cut_score || $eval > $cut_evalue ) {next;} + + $hsp_counter++; + $hit_counting{$db_element_name} = + 1; #just so we can get a value of how many hits we have... + $hash{$db_element_name}{"length"} = $db_element_length; + $hash{$db_element_name}{"hsp"}{$hsp_counter} = { + "identical" => $frac_id_db, + "conserved" => $frac_cons_db, + "start" => $hstart, + "end" => $hend, + "score" => $score, + "evalue" => $eval, + }; + $hash{$db_element_name}{"aln_prop"} = $ref_aln_prop if ($local_hsp_counter==1 &&(!$hash{$db_element_name}{"aln_prop"} || $hash{$db_element_name}{"aln_prop"}<$ref_aln_prop) ); + $hash{$db_element_name}{'align_name'} = $query_name; + } elsif ( @data && $hash_type eq "queries" && $data[0] eq "QUERY" ) { + my $query_name = $data[1]; + my $query_length = $data[2]; + $hsp_counter = $data[3]; + my $qstart = $data[6]; + my $qend = $data[7]; + my $score = $data[8]; + my $eval = $data[9]; + my $query_aln_prop = $data[10]; + my $local_hsp_counter = $data[11]; + my $local_hit_counter = $data[12] if $data[12]; + my $direction = $data[13]; + my $hit_name = $data[14]; + next if ($limit && $local_hit_counter && $limit < $local_hit_counter); + if ( $score < $cut_score ) { + next; + } # a) to prevent low scoring HSPs to be accepted - repeats; b) to allow reparsing of hash + if ( $eval > $cut_evalue ) { next; } + $query_counting{$query_name} = + 1; #just so we can get a value of how many hits we have... + $hash{$query_name}{"length"} = $query_length; + + if ($extract){ + my $start = $qstart; + my $end = $qend; + if ($end<$start){my $t=$end;$end=$start;$start=$t;} + $extr_hash{$query_name}{"start"}=$start if !$extr_hash{$query_name}{"start"} || $start < $extr_hash{$query_name}{"start"}; + $extr_hash{$query_name}{"end"}=$end if !$extr_hash{$query_name}{"end"} || $end > $extr_hash{$query_name}{"end"}; + $extr_hash{$query_name}{"direction"}= $data[13]; +#die "query $query_name has end ".$extr_hash{$query_name}{"end"}; + } + + $hash{$query_name}{"hsp"}{$hsp_counter} = { + "start" => $qstart, + "end" => $qend, + "score" => $score, + "evalue" => $eval, + }; + $hash{$query_name}{"aln_prop"} = $query_aln_prop if ($local_hsp_counter==1 &&(!$hash{$query_name}{"aln_prop"} || $hash{$query_name}{"aln_prop"}<$query_aln_prop) ); + $hash{$query_name}{'align_name'} = $hit_name; + } + } + close(HASH); + foreach my $key ( keys %hit_counting ) { $total_hit_counter++; } + foreach my $key ( keys %query_counting ) { $query_counter++; } + #my $elapsed = $timer->report("%L"); + #print "\nTime elapsed: $elapsed min.\n"; + #print LOG "\nTime elapsed: $elapsed min.\n"; + #print "Using cut-off evalue of $cut_evalue and bit-score $cut_score.\nCalculating Stats...\n"; + #print LOG "Using cut-off evalue of $cut_evalue and bit-score $cut_score.\n"; + my $hash_ref = \%hash; + return ( $hash_ref, $db_name, $db_length, $db_entries, $query_total ); +} + +# build new hash +sub build_hash ($) { + my $blastfile = shift; + my ( + %hash_db_elements, %hash_queries, $db_name, + $query_counter, $total_hit_counter, + $query_tlength_with_hits, $db_length, $db_entries, + $query_total, $global_hit_index + ); + my $hitless = int(0); + my $blast_obj = Bio::SearchIO->new( -file => $blastfile, -format => $report_style ); + my $hsp_counter=int(0); #global HSP counter + print "Building HASH...\n"; + while ( my $result = $blast_obj->next_result() ) { + + # this is the timer + $query_total++; + #if ( !$notimer && $query_total =~ /0000$/ ) { + # print $timer->report( "eta: %E min, %40b %p\r", $query_total ); + #} + if ($idfile){ + unless ( exists $ids{$result->query_name} ){ + next; + } + } + my $query_length = $result->query_length(); + $query_tlength += $query_length; + my $hitcount = $result->num_hits; + if ( $hitcount == 0 ) { $hitless++; next; } # skip queries with no hits. + + # shall we allow multiple blast reports in one file? Don't think so.... + unless ($db_name) { + $db_name = $result->database_name(); + $db_name =~ s/\s+$//; + } + unless ($db_length) { + $db_length = $result->database_letters(); + $db_length =~ s/\D//g; + } + unless ($db_entries) { + $db_entries = $result->database_entries(); + $db_entries =~ s/\D//g; + } + $query_tlength_with_hits += $query_length; + my $query_name = $result->query_name(); + $hash_queries{$query_name}{"length"} = $query_length; + my $hit_counter = int(0); + my $query_hsp_counter = int(0); + +# db_element is, essentially, each element in the BLAST database. Contrast with Query which is the elements in the query dataset. + while ( my $hit = $result->next_hit ) { + $hit_counter++; # for this query + last if ( $limit && $limit < $hit_counter ); + $hit->overlap(5); + my ($qcontigs, $scontigs) = Bio::Search::SearchUtils::tile_hsps($hit); + my $reference_length = $hit->length(); + my $tscore = $hit->bits(); + my $teval = $hit->significance(); + if ( $tscore < $cut_score ) { next; } + if ( $teval > $cut_evalue ) { next; } + if ( $hit->rank() == 1 ) { $query_counter++; } + $global_hit_index++; + my $db_element_name = $hit->name(); + my $db_element_length = $hit->length(); + my $qstrand = $hit->strand('query'); + my $rstrand = $hit->strand('hit'); + my $direction = $qstrand == $rstrand ? 'F' : 'R'; + my ($reference_aln_length,$query_aln_length); + $hash_db_elements{$db_element_name}{"length"} = $db_element_length; + if ($scontigs==1){ + my $hsp = $hit->hsp()||next; + my $hsp_rank = $hsp ->rank(); + $reference_aln_length=$hsp->length('hit'); + my $ref_aln_prop = sprintf("%.4f",$reference_aln_length/$reference_length); + $query_aln_length=$hsp->length('query'); + my $query_aln_prop = sprintf("%.4f",$query_aln_length/$query_length); + my $score = $hsp->bits(); + my $eval = $hsp->evalue(); + #if ( $hsp_rank == 1 && $score < $cut_score ) { next; } + #if ( $hsp_rank == 1 && $eval > $cut_evalue ) { next; } + $hsp_counter++; # global hsps index in whole blast report + $query_hsp_counter++; # Index for query HSPs + my ( $hstart, $hend ) = $hsp->range('hit'); + my $frac_id_db = $hsp->frac_identical('hsp'); + my $frac_cons_db = $hsp->frac_conserved('hsp'); + $frac_id_db = sprintf( "%.4f", $frac_id_db ); + $frac_cons_db = sprintf( "%.4f", $frac_cons_db ); + $hash_db_elements{$db_element_name}{"hsp"}{$hsp_counter} = { + "identical" => $frac_id_db, + "conserved" => $frac_cons_db, + "start" => $hstart, + "end" => $hend, + "score" => $score, + "evalue" => $eval + }; + $hash_queries{$query_name}{'align_name'} = $db_element_name; + $hash_queries{$query_name}{"aln_prop"} = $query_aln_prop if ( + $hsp_rank==1 && (!$hash_queries{$query_name}{"aln_prop"} || $hash_queries{$query_name}{"aln_prop"}<$query_aln_prop) + ); + $hash_db_elements{$db_element_name}{"aln_prop"} = $ref_aln_prop if ($hsp_rank==1 &&(!$hash_db_elements{$db_element_name}{"aln_prop"} || $hash_db_elements{$db_element_name}{"aln_prop"}<$ref_aln_prop) ); + $hash_db_elements{$db_element_name}{'align_name'} = $query_name; + my ( $qstart, $qend ) = $hsp->range('query'); + my $frac_id_query = "N/A"; # actually we dont want these values for queries + my $frac_cons_query = "N/A"; + if ($store_hash) { + print HASH "HIT\t$db_element_name\t$db_element_length\t$hsp_counter\t$frac_id_db\t$frac_cons_db\t$hstart\t$hend\t$score\t$eval\t$ref_aln_prop\t$hsp_rank\t$hit_counter\tF\t$query_name\n"; + print HASH "QUERY\t$query_name\t$query_length\t$hsp_counter\t$frac_id_query\t$frac_cons_query\t$qstart\t$qend\t$score\t$eval\t$query_aln_prop\t$hsp_rank\t$hit_counter\t$direction\t$db_element_name\n"; + } + }else{ + foreach my $contig (@{$scontigs}){ + $reference_aln_length+= abs($contig->{'stop'}-$contig->{'start'})+1; + my $ref_aln_prop = sprintf("%.4f",$reference_aln_length/$reference_length); + my $query_aln_prop = sprintf("%.4f",$reference_aln_length/$query_length); + my $hsp=@{$contig->{'hsps'}}[0]; + my $hsp_rank = $hsp ->rank(); + my $score = $hsp->bits(); + my $eval = $hsp->evalue(); + #if ($hsp_rank ==1 && $score < $cut_score ) { next; } + #if ($hsp_rank ==1 && $eval > $cut_evalue ) { next; } + $hsp_counter++; # global hsps index in whole blast report + $query_hsp_counter++; # Index for query HSPs + my ( $hstart, $hend ) = $hsp->range('hit'); + my $frac_id_db = $hsp->frac_identical('hsp'); + my $frac_cons_db = $hsp->frac_conserved('hsp'); + $frac_id_db = sprintf( "%.4f", $frac_id_db ); + $frac_cons_db = sprintf( "%.4f", $frac_cons_db ); + $hash_db_elements{$db_element_name}{"hsp"}{$hsp_counter} = { + "identical" => $frac_id_db, + "conserved" => $frac_cons_db, + "start" => $hstart, + "end" => $hend, + "score" => $score, + "evalue" => $eval + }; + $hash_queries{$query_name}{'align_name'} = $db_element_name; + $hash_queries{$query_name}{"aln_prop"} = $query_aln_prop if ($hsp_rank==1 &&(!$hash_queries{$query_name}{"aln_prop"} || $hash_queries{$query_name}{"aln_prop"}<$query_aln_prop) ); + $hash_db_elements{$db_element_name}{"aln_prop"} = $ref_aln_prop if ($hsp_rank==1 &&(!$hash_db_elements{$db_element_name}{"aln_prop"} || $hash_db_elements{$db_element_name}{"aln_prop"}<$ref_aln_prop) ); + $hash_db_elements{$db_element_name}{'align_name'} = $query_name; + my ( $qstart, $qend ) = $hsp->range('query'); + my $frac_id_query = "N/A"; # actually we dont want these values for queries + my $frac_cons_query = "N/A"; + if ($store_hash) { + print HASH "HIT\t$db_element_name\t$db_element_length\t$hsp_counter\t$frac_id_db\t$frac_cons_db\t$hstart\t$hend\t$score\t$eval\t$ref_aln_prop\t$hsp_rank\t$hit_counter\tF\t$query_name\n"; + print HASH "QUERY\t$query_name\t$query_length\t$hsp_counter\t$frac_id_query\t$frac_cons_query\t$qstart\t$qend\t$score\t$eval\t$query_aln_prop\t$hsp_rank\t$hit_counter\t$direction\t$db_element_name\n"; + } + } + } + } + } + if ($store_hash) { + print HASH + "#Total length of queries with hits:$query_tlength_with_hits\n"; + print HASH "#Overall query length:$query_tlength\n"; + print HASH "#Queries total:$query_total\n"; + print HASH + "#DB name:$db_name\n#DB length:$db_length\n#DB entries:$db_entries\n"; + } + foreach my $key ( keys %hash_db_elements ) { + $total_hit_counter++; + } + #my $elapsed = $timer->report("%L"); + #print +#"\nTime elapsed: $elapsed min.\nFound $query_counter queries with $total_hit_counter unique hits. $global_hit_index non-unique hits totalling $hsp_counter HSPs using cut-off evalue of $cut_evalue and bit-score $cut_score.\n$hitless queries had no hits and have been discarded.\nCalculating Stats...\n"; + #print LOG +#"\nTime elapsed: $elapsed min.\nFound $query_counter queries with $total_hit_counter unique hits. $global_hit_index non-unique hits totalling $hsp_counter HSPs using cut-off evalue of $cut_evalue and bit-score $cut_score.\n$hitless queries had no hits and have been discarded.\n"; + my $hash_ref_db_elements = \%hash_db_elements; + my $hash_ref_queries = \%hash_queries; + + # no longer return the query hash, empty query info after printing it out. + return ( $hash_ref_db_elements, $db_name, $db_length, $db_entries, + $query_total ); +} + +sub create_debug_log ($$) { + my $hash_ref = shift; + my $hash_type = shift; + print DEBUG "\n\nLooking at $hash_type\n"; + foreach my $element ( keys %$hash_ref ) { + print DEBUG "\nelement $element\t"; + foreach my $position ( sort { $a <=> $b } + ( keys %{ $hash_ref->{$element}{"pos"} } ) ) + { + my $counter = $hash_ref->{$element}{"pos"}{$position}; + print DEBUG "\n\t$position has $counter"; + } + } +} + + +sub revcomp { + my $dna = shift; + my $revcomp = reverse(uc($dna)); + $revcomp =~ tr/ACGT/TGCA/; + return $revcomp; +} + diff --git a/99.scripts/trinity_utils/util/misc/align_reads_launch_igv.pl b/99.scripts/trinity_utils/util/misc/align_reads_launch_igv.pl new file mode 100644 index 0000000..0b9b900 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/align_reads_launch_igv.pl @@ -0,0 +1,50 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib("$FindBin::Bin/../../PerlLib"); +use Pipeliner; +use Cwd; + + +my $usage = "\n\n\tusage: $0 input.fasta reads.left.[fq|fa] reads.right.[fq|fa]\n\n"; + +my $target_fa = $ARGV[0] or die $usage; +my $left_reads = $ARGV[1] or die $usage; +my $right_reads = $ARGV[2] or die $usage; + +unless ($target_fa =~ /^\//) { + $target_fa = cwd() . "/$target_fa"; +} + +main: { + + my $pipeliner = new Pipeliner('-verbose' => 2); + + $pipeliner->add_commands(new Command("samtools faidx $target_fa", "$target_fa.fai.ok")); + + my $cmd = "bowtie2-build $target_fa $target_fa"; + $pipeliner->add_commands(new Command($cmd, "$target_fa.bowtie2-build.ok")); + + + my $format = ($left_reads =~ /q$/i) ? '-q' : '-f'; + + my $alignments_file = cwd() . "/alignments.$$.bam"; + + $cmd = "set -eof pipefail; bowtie2 --no-unal -X 1000 -x $target_fa $format -1 $left_reads -2 $right_reads | samtools view -Sb - | samtools sort > $alignments_file"; + $pipeliner->add_commands(new Command($cmd, "$alignments_file.ok")); + + $pipeliner->add_commands(new Command("samtools index $alignments_file", "$alignments_file.bai.ok")); + + $pipeliner->run(); + + system("igv.sh -g $target_fa $alignments_file"); + + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/allele_simulator.pl b/99.scripts/trinity_utils/util/misc/allele_simulator.pl new file mode 100644 index 0000000..e42611d --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/allele_simulator.pl @@ -0,0 +1,113 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; +use List::Util qw(min max); +use Data::Dumper; + +my $usage = "usage: $0 transcriptome.fasta polymorphRatePercentage\n\n"; + +my $transcriptome = $ARGV[0] or die $usage; +my $poly_rate = $ARGV[1] or die $usage; + +if ($poly_rate < 1) { + print STDERR "\n\n** WARNING: polymorphRatePercentage expects a percentage, and your input value is quite small: $poly_rate ** \n\n"; +} + +my %mutations; + +main: { + + my $fasta_reader = new Fasta_reader($transcriptome); + + my $counter = 0; + + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + my $sequence = $seq_obj->get_sequence(); + + my $num_snps = int($poly_rate/100 * length($sequence) + 0.5); + + if ($num_snps < 1) { next; } # exclude since not helpful here. + + &print_fasta("aleA;$acc", $sequence); + + $sequence = &mutate($sequence, $num_snps); + + &print_fasta("aleB;$acc", $sequence); + + $counter++; + print STDERR "\r[$counter] "; + + } + print STDERR "\n\n"; + + #print STDERR Dumper(\%mutations); + + exit(0); +} + + +#### +sub print_fasta { + my ($acc, $sequence) = @_; + + $sequence =~ s/(\S{60})/$1\n/g; + + chomp $sequence; + + print ">$acc\n$sequence\n"; + + return; +} + + +#### +sub mutate { + my ($sequence, $num_snps) = @_; + + my %seen; + + my @chars = qw(G A T C); + + + my @seq = split(//, uc $sequence); + + for (1..$num_snps) { + + my $pos; # select unique sites (sampling w/o replacement) + + do { + $pos = int(rand(length($sequence))); + + } while ($seen{$pos}); + + $seen{$pos} = 1; + + my $nuc = uc $seq[$pos]; + + my @others = grep { $_ ne $nuc } @chars; + + my $substitution = $others[ int(rand(scalar(@others))) ]; + + #print STDERR "$nuc -> $substitution\n"; + + #$mutations{"$nuc,$substitution"}++; + + $seq[$pos] = lc $substitution; + + } + + $sequence = join("", @seq); + + return($sequence); +} + + + + diff --git a/99.scripts/trinity_utils/util/misc/alt_GG_read_partitioning_JCornish/genwig.sh b/99.scripts/trinity_utils/util/misc/alt_GG_read_partitioning_JCornish/genwig.sh new file mode 100644 index 0000000..fd0540c --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/alt_GG_read_partitioning_JCornish/genwig.sh @@ -0,0 +1,80 @@ +#!/bin/bash +sam2wig(){ + chrlen=$1 + chr=$2 + infi=$3 + nosingle=$4 + minins=$5 + maxins=$6 + threads=$7 + + outfi="$infi.$chr.wig" + tmp="$infi.$chr.sam" + + + samtools view -@ $threads $infi $chr | \ + ./genwig2.py $chrlen $chr - $outfi $nosingle $minins $maxins + echo "completed: $chr" +} +export -f sam2wig + +while getopts ":b:m:n:o:p:q:" opt; do + case $opt in + b) + if [ ! -f $OPTARG ] + then + echo "ERROR: Unable to find input bam $OPTARG" >&2 + exit 1 + else + inbam="$OPTARG" + fi + ;; + m) + maxins=$OPTARG + ;; + n) + minins=$OPTARG + ;; + o) + if [ ! -d $OPTARG ]; + then + echo "ERROR: Unable to find output directory $OPTARG" >&2 + exit 1 + else + outdir="$OPTARG" + fi + ;; + p) + nproc=$OPTARG + ;; + q) + nosingle=$OPTARG + ;; + \?) + echo "ERROR: Unknown argument $opt" >&2 + exit 1 + ;; + esac +done + +#dump chromosome info from bam index +chrinfo="$inbam.chr" +#sort is required to preserve order between trinity and here +samtools idxstats $inbam | grep -v "^*" | cut -f1,2 | sort > $chrinfo + +#generate wigs for segs/chrs +parallel -j $nproc --xapply \ +sam2wig {1} {2} $inbam $nosingle $minins $maxins $nproc \ +::: `cut -f2 $chrinfo` ::: `cut -f1 $chrinfo` + +#cat to final wig and remove file +if [ -f "$inbam.wig" ] +then + rm "$inbam.wig" +fi + +while read chr len +do + cat "$inbam.$chr.wig" >> "$inbam.wig" + rm "$inbam.$chr.wig" +done < $chrinfo diff --git a/99.scripts/trinity_utils/util/misc/alt_GG_read_partitioning_JCornish/genwig2.py b/99.scripts/trinity_utils/util/misc/alt_GG_read_partitioning_JCornish/genwig2.py new file mode 100644 index 0000000..d44b397 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/alt_GG_read_partitioning_JCornish/genwig2.py @@ -0,0 +1,95 @@ +#!/usr/bin/env python3 +import sys +import re +import numpy as np + +#regex for cigar +re_split_cigar = re.compile('(\d+[MDI])(?=[0-9]|$)') + +#SAM INDEXES +idx_qname = 0 +idx_flag = 1 +idx_rname = 2 +idx_pos = 3 +idx_mapq = 4 +idx_cigar = 5 +idx_rnext = 6 +idx_pnext = 7 +idx_tlen = 8 +idx_seq = 9 +idx_qual = 10 + +def write_err(msg, exit=False, status=1): + sys.stderr.write(msg) + if exit: + sys.exit(status) + +if __name__ == '__main__': + chrlen = int(sys.argv[1]) + _chr = sys.argv[2] + infile = sys.argv[3] + outfile = sys.argv[4] + nosingle = int(sys.argv[5]) + minins = int(sys.argv[6]) + maxins = int(sys.argv[7]) + + #open input sam + if infile == "-": + insam = sys.stdin + + else: + try: + insam = open(infile, 'r') + except IOError: + write_err("ERROR: Unable to read input sam file\n", exit=True) + + #get file for writing + if outfile == "-": + outwig = sys.stdout + + else: + try: + outwig = open(outfile, 'w') + except IOError: + write_err("ERROR: Unable to open output file\n", exit=True) + + #start generating wig + wig = np.zeros(chrlen, dtype = np.uint64) + + for row in insam: + entry = row.split('\t') + start = int(entry[idx_pos]) - 1 + tlen = int(entry[idx_tlen]) + atlen = abs(tlen) + + #entry = gen_sam(row) + #start = entry.pos - 1 + #atlen = abs(entry.tlen) + if atlen > 0 and atlen >= minins and atlen <= maxins: + if tlen > 0: + wig[start:(start + tlen)] += 1 + else: + continue + else: + cigars = re.split(re_split_cigar, entry[idx_cigar]) + cigars = [x for x in cigars if x != ''] + _len = 0 + + for c in cigars: + if 'M' in c: + _len += int(c.split('M')[0]) + elif 'D' in c: + _len += int(c.split('D')[0]) + elif 'I' in c: + continue + _len += int(c.split('I')[0]) + else: + continue + wig[start:(start + _len)] += 1 + + + outwig.write("variableStep chrom=" + _chr + "\n") + for i in range(0, chrlen): + outwig.write(str(i + 1) + "\t" + str(wig[i]) + "\n") + + diff --git a/99.scripts/trinity_utils/util/misc/altsplice_simulation_toolkit/sim_single_bubble.pl b/99.scripts/trinity_utils/util/misc/altsplice_simulation_toolkit/sim_single_bubble.pl new file mode 100644 index 0000000..7d60e86 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/altsplice_simulation_toolkit/sim_single_bubble.pl @@ -0,0 +1,40 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use lib ($ENV{EUK_MODULES}); +use Fasta_reader; + +my $usage = "usage: $0 targets.fasta\n\n"; + +my $target_fasta_file = $ARGV[0] or die $usage; + + +main: { + + my $fasta_reader = new Fasta_reader($target_fasta_file); + + my %seqs = $fasta_reader->retrieve_all_seqs_hash(); + + foreach my $acc (keys %seqs) { + my $sequence = $seqs{$acc}; + + my $seq_len = length($sequence); + if ($seq_len < 500) { next; } + + my $bubble_missing_seq = $sequence; + $bubble_missing_seq = substr($bubble_missing_seq, 0, 200) . substr($bubble_missing_seq, 350); + + my $new_gene_acc = $acc; + $new_gene_acc =~ s/\W/_/g; + + print ">isoA-$new_gene_acc;$new_gene_acc\n$sequence\n" + . ">isoB-$new_gene_acc;$new_gene_acc\n$bubble_missing_seq\n"; + + } + + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/misc/analyze_blastPlus_topHit_coverage.annotate_details_w_FL_info.pl b/99.scripts/trinity_utils/util/misc/analyze_blastPlus_topHit_coverage.annotate_details_w_FL_info.pl new file mode 100644 index 0000000..d74b3ec --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/analyze_blastPlus_topHit_coverage.annotate_details_w_FL_info.pl @@ -0,0 +1,128 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + + +my $help_flag; + +my $usage = <<__EOUSAGE__; + +################################################################################################# +# +# Required: +# +# --details compreh_init_build.details filename +# --blast blast.outfmt6.w_pct_hit_length filename +# +# Optional: +# +# --min_pct_hit_len min length of alignment coverage of hit (default: 80) +# --pasa_validations incorporate notes from pasa's alignment validations file (alignment.validations.out) +# +################################################################################################# + + +__EOUSAGE__ + + ; + + +my $compreh_details_file; +my $blast_file; +my $min_pct_hit_len = 80; +my $pasa_validations_file; + +&GetOptions ( 'h' => \$help_flag, + 'details=s' => \$compreh_details_file, + 'blast=s' => \$blast_file, + 'min_pct_hit_len=f' => \$min_pct_hit_len, + 'pasa_validations=s' => \$pasa_validations_file, + + ); + + +if ($help_flag) { + die $usage; +} + +unless ($compreh_details_file && $blast_file) { + die $usage; +} + +my %pasa_validations_info; +if ($pasa_validations_file) { + + open (my $fh, $pasa_validations_file) or die $!; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my $acc = $x[1]; + my $note = $x[14]; + if ($note) { + $pasa_validations_info{$acc} .= $note . "; "; + } + } + close $fh; + +} + +my %FL_mappings; + +{ + open (my $fh, $blast_file) or die $!; + while (<$fh>) { + if (/^\#/) { next; } + + chomp; + my @x = split(/\t/); + my $pct_hit_len = $x[13]; + if ($pct_hit_len < $min_pct_hit_len) { + next; + } + + my $query = $x[0]; + my $hit = $x[1]; + my $annot = $x[14]; + my $Evalue = $x[10]; + + $FL_mappings{$query} = join("\t", $hit, $annot, $Evalue, $pct_hit_len); + + } + close $fh; +} + +my $prev_gene = ""; +open (my $fh, $compreh_details_file) or die $!; +while (<$fh>) { + chomp; + my @x = split(/\t/); + my $gene = $x[0]; + my $acc = $x[1]; + if (my $mappings = $FL_mappings{$acc}) { + $mappings =~ s/\t/ /g; + push (@x, $mappings); + } + else { + push (@x, ""); + } + if (my $validation_note = $pasa_validations_info{$acc}) { + push (@x, "validation_note: $validation_note"); + } + + + if ($gene ne $prev_gene) { + print "\n"; + } + $prev_gene = $gene; + + print join("\t", @x) . "\n"; +} + + + +exit(0); + + diff --git a/99.scripts/trinity_utils/util/misc/analyze_blastPlus_topHit_coverage.by_prioritized_compreh_category.pl b/99.scripts/trinity_utils/util/misc/analyze_blastPlus_topHit_coverage.by_prioritized_compreh_category.pl new file mode 100644 index 0000000..1855cc2 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/analyze_blastPlus_topHit_coverage.by_prioritized_compreh_category.pl @@ -0,0 +1,233 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + + +my $help_flag; + +my $usage = <<__EOUSAGE__; + +################################################################################################# +# +# Required: +# +# --details compreh_init_build.details filename +# --blast blast.outfmt6.w_pct_hit_length filename +# +# Optional: +# +# --min_pct_detail_total minimum percent of total assignments to capture for a given detail category (default: 1.0) +# --min_pct_hit_len min length of alignment coverage of hit (default: 80) +# --max_Evalue default (1e-20); +# +################################################################################################# + + +__EOUSAGE__ + + ; + + +my $compreh_details_file; +my $blast_file; +my $min_pct_hit_len = 80; +my $min_pct_detail_total = 1.0; +my $max_Evalue = 1e-20; + + +&GetOptions ( 'h' => \$help_flag, + 'details=s' => \$compreh_details_file, + 'blast=s' => \$blast_file, + 'min_pct_hit_len=f' => \$min_pct_hit_len, + 'min_pct_detail_total=f' => \$min_pct_detail_total, + 'max_Evalue=f' => \$max_Evalue, + + ); + + +if ($help_flag) { + die $usage; +} + +unless ($compreh_details_file && $blast_file) { + die $usage; +} + + + +my %priorities = ( 'pasa' => 1, + 'InvalidQualityAlignment' => 2, + 'PoorAlignment' => 3, + 'TDN' => 4, + ); + +my %acc_to_detail; +{ + open (my $fh, $compreh_details_file) or die $!; + while (<$fh>) { + chomp; + my ($gene, $trans, $detail) = split(/\t/); + my ($detail_prefix, $rest) = split(/_/, $detail); + $acc_to_detail{$trans} = $detail_prefix; + } + close $fh; +} + + +my %FL_mappings; # hit acc => detail +my %hit_to_OS; +my %hit_to_query; + +{ + open (my $fh, $blast_file) or die $!; + while (<$fh>) { + if (/^\#/) { next; } + + chomp; + my @x = split(/\t/); + my $pct_hit_len = $x[13]; + if ($pct_hit_len < $min_pct_hit_len) { + next; + } + + my $evalue = $x[10]; + if ($evalue > $max_Evalue) { next; } + + + my $query = $x[0]; + my $hit = $x[1]; + my $annot = $x[14]; + + my $query_detail = $acc_to_detail{$query}; + + + if ($annot =~ /OS=(\S+)/) { + my $os = $1; + if ($hit_to_OS{$hit}) { + + ## see if this has higher priority + my $prev_detail = $FL_mappings{$hit}; + if ($priorities{$prev_detail} > $priorities{$query_detail}) { + $FL_mappings{$hit} = $query_detail; # assign the lower value, meaning higher priority here. + $hit_to_query{$hit} = $query; + } + + } + else { + $hit_to_OS{$hit} = $os; + $FL_mappings{$hit} = $query_detail; + $hit_to_query{$hit} = $query; + } + } + else { + print STDERR "Error, no OS specified for $annot"; + } + } + close $fh; + +} + + +## reformat data structure to obtain detail -> OS -> count +my %data; +my %detail_count_totals; +my %os_total_counts; +my %detail_to_query_list; + +{ + foreach my $hit (keys %FL_mappings) { + + my $detail = $FL_mappings{$hit}; + my $os = $hit_to_OS{$hit}; + + my $query = $hit_to_query{$hit}; + + $data{$detail}->{$os}++; + $detail_count_totals{$detail}++; + + push (@{$detail_to_query_list{$detail}}, { query => $query, + os => $os, + }); + } + + open (my $ofh, ">prioritized_query_to_os_summary.len$min_pct_hit_len.E$max_Evalue.txt") or die $!; + + ## report + foreach my $detail (sort {$priorities{$a}<=>$priorities{$b}} keys %priorities) { + + my $detail_total = $detail_count_totals{$detail}; + + print "DETAIL: $detail Total: $detail_total\n"; + + my $os_counts_href = $data{$detail}; + + foreach my $os (reverse sort {$os_counts_href->{$a}<=>$os_counts_href->{$b}} keys %$os_counts_href) { + + my $count = $os_counts_href->{$os}; + + my $percent = sprintf("%.2f", $count/$detail_total*100); + + if ($percent >= $min_pct_detail_total) { + + print "$os\t$count\t$percent%\n"; + + $os_total_counts{$os}+= $count; + } + + + + } + + print "\n"; + + my @entries = @{$detail_to_query_list{$detail}}; + + foreach my $entry (@entries) { + + my $query = $entry->{query}; + my $os = $entry->{os}; + print $ofh join("\t", $detail, $query, $os) . "\n"; + } + + } + close $ofh; +} + +## output matrix +my @os_rows = reverse sort {$os_total_counts{$a}<=>$os_total_counts{$b}} keys %os_total_counts; +my @details = sort {$priorities{$a}<=>$priorities{$b}} keys %priorities; + + +foreach my $type ('counts', 'percentages') { + + print "\n\n"; + print "** Matrix of $type **\n\n"; + + print "#\t" . join("\t", @details) . "\n"; + foreach my $os (@os_rows) { + print "$os"; + foreach my $detail (@details) { + my $count = $data{$detail}->{$os} || 0; + + my $detail_total = $detail_count_totals{$detail}; + my $percent = sprintf("%.2f", $count/$detail_total*100); + if ($type eq 'counts') { + print "\t$count"; + } + else { + print "\t$percent"; + } + } + #my $os_total = $os_total_counts{$os}; + #print "\t\ttotal: $os_total\n"; + print "\n"; + } + print "\n\n"; +} + +exit(0); + + diff --git a/99.scripts/trinity_utils/util/misc/analyze_blastPlus_topHit_coverage.extract_OS.pl b/99.scripts/trinity_utils/util/misc/analyze_blastPlus_topHit_coverage.extract_OS.pl new file mode 100644 index 0000000..0248fb3 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/analyze_blastPlus_topHit_coverage.extract_OS.pl @@ -0,0 +1,102 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + +my $usage = <<__EOUSAGE__; + +######################################################################################### +# +# Required: +# +# --blast_outfmt6_w_pct_hit_length blast.outfmt6.w_pct_hit_length +# (results from running analyze_blastPlus_tophat_coverage.pl) +# +# Optional: +# +# --min_pct_hit_length minimum percent hit length to be included in analysis. (default: 20) +# +# --min_pct_species_report minimum percent of total species content to +# be reported in output (default: 1.0) +# +########################################################################################### + + +__EOUSAGE__ + + ; + +my $file; +my $min_pct_hit_len = 200; +my $min_pct_species_report = 1.0; + +my $help_flag; + +&GetOptions( 'blast_outfmt6_w_pct_hit_length=s' => \$file, + 'min_pct_hit_length=i' => \$min_pct_hit_len, + 'min_pct_species_report=f' => \$min_pct_species_report, + + 'help|h' => \$help_flag, + ); + + +if ($help_flag) { + die $usage; +} + +if (@ARGV) { + die "Error, didn't parse parameters: @ARGV"; +} + +unless ($file) { + die $usage; +} + + +my $total = 0; +my %OS_counter; + +my $prev_acc = ""; + +open (my $fh, $file) or die $!; +while (<$fh>) { + if (/^\#/) { next; } + + chomp; + my @x = split(/\t/); + + my $acc = $x[0]; + if ($acc eq $prev_acc) { next; } + + + my $pct_hit_len = $x[13]; + if ($pct_hit_len < $min_pct_hit_len) { + next; + } + + my $annot = $x[14]; + + if ($annot =~ /OS=(\S+)/) { + my $os = $1; + $total++; + $OS_counter{$os}++; + + $prev_acc = $acc; + + } +} +close $fh; + +foreach my $os (sort {$OS_counter{$b} <=> $OS_counter{$a}} keys %OS_counter) { + + my $count = $OS_counter{$os}; + my $pct = sprintf("%.2f", $count/$total*100); + + print join("\t", $os, $count, "$pct%") . "\n" if $pct >= $min_pct_species_report; +} + +exit(0); + + diff --git a/99.scripts/trinity_utils/util/misc/analyze_blastPlus_topHit_coverage.org_matrix.pl b/99.scripts/trinity_utils/util/misc/analyze_blastPlus_topHit_coverage.org_matrix.pl new file mode 100644 index 0000000..14fa977 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/analyze_blastPlus_topHit_coverage.org_matrix.pl @@ -0,0 +1,57 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 (count|percent) A.org_rep B.org_rep ...\n\n"; + +my $type = shift @ARGV; + +my @org_rep_files = @ARGV or die $usage; + +unless ($type eq 'count' || $type eq 'percent') { die $usage; } + + +main: { + + my %matrix; + my %species_to_total_percentage; + + foreach my $org_rep_file (@org_rep_files) { + open (my $fh, $org_rep_file) or die $!; + while (<$fh>) { + chomp; + my ($species, $count, $percent) = split(/\t/); + $percent =~ s/\%//; + + my $value = $percent; + if ($type eq 'count') { + $value = $count; + } + + $matrix{$species}->{$org_rep_file} = $value; + + $species_to_total_percentage{$species} += $value; + } + close $fh; + } + + + my @species = reverse sort {$species_to_total_percentage{$a} <=> $species_to_total_percentage{$b}} keys %species_to_total_percentage; + + print "#\t" . join("\t", @org_rep_files) . "\n"; + + foreach my $specie (@species) { + print "$specie"; + foreach my $org (@org_rep_files) { + + my $count = $matrix{$specie}->{$org} || 0; + print "\t$count"; + } + print "\n"; + } + + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/misc/average.pl b/99.scripts/trinity_utils/util/misc/average.pl new file mode 100644 index 0000000..09256e2 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/average.pl @@ -0,0 +1,39 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use BHStats; + +my $count = 0; +my $sum = 0; +my @values; + +while () { + chomp; + $sum += $_; + $count++; + push (@values, $_); +} + +@values = sort {$a<=>$b} @values; + +if ($count) { + my $average = ($sum/$count); + my $median_pos = int ($count/2); + print "\n"; + print "MIN: " . $values[0] . "\n"; + print "MAX: " . $values[$#values] . "\n"; + print "Sum: $sum\n"; + printf ("Average: %.2f\n", $average); + print "Median: " . BHStats::median(@values) . "\n"; + my $stdev = BHStats::stDev(@values); + printf ("stDev from Average: %.2f\n", $stdev); + + my $geoMean = &BHStats::geometric_mean(@values); + if ($geoMean) { + printf ("geoMean: %.2f\n", $geoMean); + } +} + diff --git a/99.scripts/trinity_utils/util/misc/bam_gene_tests/extract_bam_reads_per_target_gene.pl b/99.scripts/trinity_utils/util/misc/bam_gene_tests/extract_bam_reads_per_target_gene.pl new file mode 100644 index 0000000..58fc788 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/bam_gene_tests/extract_bam_reads_per_target_gene.pl @@ -0,0 +1,264 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use File::Path; +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use File::Basename; + +use FindBin; +use lib ("$FindBin::RealBin/../../../PerlLib"); + +use Nuc_translator; +use SAM_reader; +use SAM_entry; + + +my $usage = <<__EOUSAGE__; + +################################################################ +# +# Required: +# +# --coord_sorted_SAM coordinate-sorted SAM file +# +# Options: +# +# --SS_lib_type [SS_lib_type=F,R,FR,RF] +# +# --parts_per_directory default: 100 +# --min_reads_per_partition default: 10 +# +################################################################# + +__EOUSAGE__ + + ; + + +my $alignments_sam; +my $SS_lib_type; + +my $PARTS_PER_DIR = 100; + +my $MIN_READS_PER_PARTITION = 10; + +my $help_flag = 0; + +&GetOptions ( + 'help|h' => \$help_flag, + + 'coord_sorted_SAM=s' => \$alignments_sam, + 'SS_lib_type=s' => \$SS_lib_type, + 'parts_per_directory=i' => \$PARTS_PER_DIR, + 'min_reads_per_partition=i' => \$MIN_READS_PER_PARTITION, + ); + +if ($help_flag) { + die $usage; +} + +unless ($alignments_sam) { + die $usage; +} + + +main: { + + + my $partitions_dir = "ReadPartitions"; + unless (-d $partitions_dir) { + mkdir ($partitions_dir) or die "Error, cannot mkdir $partitions_dir"; + } + open (my $track_fh, ">$partitions_dir.listing") or die $!; + + my $current_partition = undef; + + my $ofh; + + my $sam_ofh; + + my $partition_counter = 0; + + my $part_file = ""; + my $sam_part_file = ""; + my $read_counter = 0; + + my $current_scaff = ""; + + my $sam_reader = new SAM_reader($alignments_sam); + while (my $sam_entry = $sam_reader->get_next()) { + + my $acc = $sam_entry->reconstruct_full_read_name(); + my $scaff = $sam_entry->get_scaffold_name(); + + my ($trans, $gene) = split(/;/, $scaff); + $scaff = $gene; + + + next if $scaff eq '*'; + + my $seq = $sam_entry->get_sequence(); + + my $read_name = $sam_entry->get_read_name(); # raw from sam file + if ($acc !~ /\/[12]$/ && $read_name =~ /\/[12]$/) { + $acc = $read_name; + } + + + my $aligned_strand = $sam_entry->get_query_strand(); + my $opposite_strand = ($aligned_strand eq '+') ? '-' : '+'; + + if ($aligned_strand eq '-') { + # restore to actual sequenced bases + $seq = &reverse_complement($seq); + } + + if ($SS_lib_type) { + ## got SS data + + my $transcribed_orient; + + if (! $sam_entry->is_paired()) { + if ($SS_lib_type !~ /^(F|R)$/) { + confess "Error, read is not paired but SS_lib_type set to paired: $SS_lib_type\nread:\n$_"; + } + + if ($SS_lib_type eq "R") { + $seq = &reverse_complement($seq); + } + } + + else { + ## Paired reads. + if ($SS_lib_type !~ /^(FR|RF)$/) { + confess "Error, read is paired but SS_lib_type set to unpaired: $SS_lib_type\nread:\n$_"; + } + + my $first_in_pair = $sam_entry->is_first_in_pair(); + if ( ($first_in_pair && $SS_lib_type eq "RF") + || + ( (! $first_in_pair) && $SS_lib_type eq "FR") + ) { + $seq = &reverse_complement($seq); + } + } + } + + + my $new_partition_flag = 0; + + ## prime ordered partitions if first entry or if switching scaffolds. + if ($scaff ne $current_scaff) { + $new_partition_flag = 1; + $current_scaff = $scaff; + } + + + if ($new_partition_flag) { + + close $ofh if $ofh; + $ofh = undef; + + close $sam_ofh if $sam_ofh; + $sam_ofh = undef; + + if ($read_counter < $MIN_READS_PER_PARTITION) { + # delete these read files. + #print STDERR "-- too few reads ($read_counter), removing partition: $part_file\n"; + unlink($part_file, $sam_part_file); + $partition_counter--; + } + + $read_counter = 0; + + $partition_counter++; + } + + + # may need to start a new ofh for this partition if not already established. + unless ($ofh) { + # create new one. + my $file_part_count = int($partition_counter/$PARTS_PER_DIR); + my $outdir = "$partitions_dir/$file_part_count"; + $outdir =~ s/[\;\|]/_/g; + + mkpath($outdir) if (! -d $outdir); + unless (-d $outdir) { + die "Error, cannot mkdpath $outdir"; + } + + my $scaff_file_name = $current_scaff; + $scaff_file_name =~ s/\W/_/g; + + $part_file = "$outdir/$scaff_file_name.reads"; + open ($ofh, ">>$part_file") or die "Error, cannot write ot $part_file"; + print STDERR "-writing to $part_file\n"; + + $sam_part_file = "$outdir/$scaff_file_name.sam"; + open ($sam_ofh, ">>$sam_part_file") or die "Error, cannot open $sam_part_file"; + } + + + # write to partition + print $ofh ">$acc\n$seq\n"; + print $sam_ofh join("\t", $sam_entry->get_fields()) . "\n";; + $read_counter++; + + + + } + close $track_fh; + + close $ofh if $ofh; + close $sam_ofh if $sam_ofh; + + exit(0); +} + + + +#### +sub parse_partitions { + my ($partitions_file) = @_; + + my %scaff_to_parts; + + print STDERR "// parsing paritions.\n"; + my $counter = 0; + + open (my $fh, $partitions_file) or die "Error, cannot open file $partitions_file"; + while (<$fh>) { + chomp; + if (/^\#/) { next; } + unless (/\w/) { next; } + + $counter++; + print STDERR "\r[$counter] " if $counter % 100 == 0; + + my @x = split(/\t/); + + my $scaff = $x[0]; + my $lend = $x[3]; + my $rend = $x[4]; + my $orient = $x[6]; + + push (@{$scaff_to_parts{$scaff}}, { scaff => $scaff, + lend => $lend, + rend => $rend, } ); + + } + print STDERR "\r[$counter] "; + + close $fh; + + # should be sorted, but let's just be sure: + foreach my $scaff (keys %scaff_to_parts) { + @{$scaff_to_parts{$scaff}} = sort {$a->{lend}<=>$b->{lend}} @{$scaff_to_parts{$scaff}}; + } + + return(%scaff_to_parts); +} + + diff --git a/99.scripts/trinity_utils/util/misc/bam_gene_tests/extract_bam_reads_per_target_transcript.pl b/99.scripts/trinity_utils/util/misc/bam_gene_tests/extract_bam_reads_per_target_transcript.pl new file mode 100644 index 0000000..e655397 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/bam_gene_tests/extract_bam_reads_per_target_transcript.pl @@ -0,0 +1,259 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use File::Path; +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use File::Basename; + +use FindBin; +use lib ("$FindBin::RealBin/../../../PerlLib"); + +use Nuc_translator; +use SAM_reader; +use SAM_entry; + + +my $usage = <<__EOUSAGE__; + +################################################################ +# +# Required: +# +# --coord_sorted_SAM coordinate-sorted SAM file +# +# Options: +# +# --SS_lib_type [SS_lib_type=F,R,FR,RF] +# +# --parts_per_directory default: 100 +# --min_reads_per_partition default: 10 +# +################################################################# + +__EOUSAGE__ + + ; + + +my $alignments_sam; +my $SS_lib_type; + +my $PARTS_PER_DIR = 100; + +my $MIN_READS_PER_PARTITION = 10; + +my $help_flag = 0; + +&GetOptions ( + 'help|h' => \$help_flag, + + 'coord_sorted_SAM=s' => \$alignments_sam, + 'SS_lib_type=s' => \$SS_lib_type, + 'parts_per_directory=i' => \$PARTS_PER_DIR, + 'min_reads_per_partition=i' => \$MIN_READS_PER_PARTITION, + ); + +if ($help_flag) { + die $usage; +} + +unless ($alignments_sam) { + die $usage; +} + + +main: { + + + my $partitions_dir = "ReadPartitions"; + unless (-d $partitions_dir) { + mkdir ($partitions_dir) or die "Error, cannot mkdir $partitions_dir"; + } + open (my $track_fh, ">$partitions_dir.listing") or die $!; + + my $current_partition = undef; + + my $ofh; + + my $sam_ofh; + + my $partition_counter = 0; + + my $part_file = ""; + my $sam_part_file = ""; + my $read_counter = 0; + + my $current_scaff = ""; + + my $sam_reader = new SAM_reader($alignments_sam); + while (my $sam_entry = $sam_reader->get_next()) { + + my $acc = $sam_entry->reconstruct_full_read_name(); + my $scaff = $sam_entry->get_scaffold_name(); + next if $scaff eq '*'; + + my $seq = $sam_entry->get_sequence(); + + my $read_name = $sam_entry->get_read_name(); # raw from sam file + if ($acc !~ /\/[12]$/ && $read_name =~ /\/[12]$/) { + $acc = $read_name; + } + + + my $aligned_strand = $sam_entry->get_query_strand(); + my $opposite_strand = ($aligned_strand eq '+') ? '-' : '+'; + + if ($aligned_strand eq '-') { + # restore to actual sequenced bases + $seq = &reverse_complement($seq); + } + + if ($SS_lib_type) { + ## got SS data + + my $transcribed_orient; + + if (! $sam_entry->is_paired()) { + if ($SS_lib_type !~ /^(F|R)$/) { + confess "Error, read is not paired but SS_lib_type set to paired: $SS_lib_type\nread:\n$_"; + } + + if ($SS_lib_type eq "R") { + $seq = &reverse_complement($seq); + } + } + + else { + ## Paired reads. + if ($SS_lib_type !~ /^(FR|RF)$/) { + confess "Error, read is paired but SS_lib_type set to unpaired: $SS_lib_type\nread:\n$_"; + } + + my $first_in_pair = $sam_entry->is_first_in_pair(); + if ( ($first_in_pair && $SS_lib_type eq "RF") + || + ( (! $first_in_pair) && $SS_lib_type eq "FR") + ) { + $seq = &reverse_complement($seq); + } + } + } + + + my $new_partition_flag = 0; + + ## prime ordered partitions if first entry or if switching scaffolds. + if ($scaff ne $current_scaff) { + $new_partition_flag = 1; + $current_scaff = $scaff; + } + + + if ($new_partition_flag) { + + close $ofh if $ofh; + $ofh = undef; + + close $sam_ofh if $sam_ofh; + $sam_ofh = undef; + + if ($read_counter < $MIN_READS_PER_PARTITION) { + # delete these read files. + #print STDERR "-- too few reads ($read_counter), removing partition: $part_file\n"; + unlink($part_file, $sam_part_file); + $partition_counter--; + } + + $read_counter = 0; + + $partition_counter++; + } + + + # may need to start a new ofh for this partition if not already established. + unless ($ofh) { + # create new one. + my $file_part_count = int($partition_counter/$PARTS_PER_DIR); + my $outdir = "$partitions_dir/$file_part_count"; + $outdir =~ s/[\;\|]/_/g; + + mkpath($outdir) if (! -d $outdir); + unless (-d $outdir) { + die "Error, cannot mkdpath $outdir"; + } + + my $scaff_file_name = $current_scaff; + $scaff_file_name =~ s/\W/_/g; + + $part_file = "$outdir/$scaff_file_name.reads"; + open ($ofh, ">$part_file") or die "Error, cannot write ot $part_file"; + print STDERR "-writing to $part_file\n"; + + $sam_part_file = "$outdir/$scaff_file_name.sam"; + open ($sam_ofh, ">$sam_part_file") or die "Error, cannot open $sam_part_file"; + } + + + # write to partition + print $ofh ">$acc\n$seq\n"; + print $sam_ofh join("\t", $sam_entry->get_fields()) . "\n";; + $read_counter++; + + + + } + close $track_fh; + + close $ofh if $ofh; + close $sam_ofh if $sam_ofh; + + exit(0); +} + + + +#### +sub parse_partitions { + my ($partitions_file) = @_; + + my %scaff_to_parts; + + print STDERR "// parsing paritions.\n"; + my $counter = 0; + + open (my $fh, $partitions_file) or die "Error, cannot open file $partitions_file"; + while (<$fh>) { + chomp; + if (/^\#/) { next; } + unless (/\w/) { next; } + + $counter++; + print STDERR "\r[$counter] " if $counter % 100 == 0; + + my @x = split(/\t/); + + my $scaff = $x[0]; + my $lend = $x[3]; + my $rend = $x[4]; + my $orient = $x[6]; + + push (@{$scaff_to_parts{$scaff}}, { scaff => $scaff, + lend => $lend, + rend => $rend, } ); + + } + print STDERR "\r[$counter] "; + + close $fh; + + # should be sorted, but let's just be sure: + foreach my $scaff (keys %scaff_to_parts) { + @{$scaff_to_parts{$scaff}} = sort {$a->{lend}<=>$b->{lend}} @{$scaff_to_parts{$scaff}}; + } + + return(%scaff_to_parts); +} + + diff --git a/99.scripts/trinity_utils/util/misc/bam_gene_tests/harvest_transcripts.pl b/99.scripts/trinity_utils/util/misc/bam_gene_tests/harvest_transcripts.pl new file mode 100644 index 0000000..091381b --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/bam_gene_tests/harvest_transcripts.pl @@ -0,0 +1,40 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use File::Basename; + +my $usage = "usage: $0 bfly_fasta_files.list.file\n\n"; + +my $bfly_list_file = $ARGV[0] or die $usage; + + +main: { + + open (my $fh, $bfly_list_file) or die $!; + while (<$fh>) { + chomp; + + my $file = $_; + + my $base = basename($file); + my @pts = split(/\./, $base); + my $core = shift @pts; + + open (my $fh2, $file) or die "Error, cannot open file $file"; + while (<$fh2>) { + + if (/>/) { + s/>/>$core-/; + } + print; + } + close $fh2; + } + close $fh; + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/bam_gene_tests/write_trin_cmds.pl b/99.scripts/trinity_utils/util/misc/bam_gene_tests/write_trin_cmds.pl new file mode 100644 index 0000000..8e2edfb --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/bam_gene_tests/write_trin_cmds.pl @@ -0,0 +1,83 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + +my $usage = <<__EOUSAGE__; + +#################################################################################### +# +# usage: $0 --reads_list_file --out_token [Trinity params] +# +# Required: +# +# --reads_list_file file containing list of filenames corresponding +# to the reads.fasta +# +# --out_token token added to the output file name +# +##################################################################################### + +# Example: +# +# write_trin_cmds.pl --reads_list_file ReadPartitions.listing --out_token origbfly --SS_lib_type F --full_cleanup_ET --CPU 1 --bfly_jar ~/SVN/trinityrnaseq/trunk/Butterfly/Butterfly.jar --JM 1G --seqType fa + + +__EOUSAGE__ + + ; + + +my $reads_file; +my $help_flag; +my $out_token = ""; + + +&GetOptions ( + + 'reads_list_file=s' => \$reads_file, + 'h' => \$help_flag, + + 'out_token=s' => \$out_token, + + + ); + +my @TRIN_ARGS = @ARGV; + +if ($help_flag) { + die $usage; +} + +unless ($reads_file && -s $reads_file) { + die $usage; +} + +unless ($out_token) { + die $usage; +} + +my $trin_args = join(" ", @TRIN_ARGS); + + +open (my $fh, $reads_file) or die "Error, cannot open file $reads_file"; +while (<$fh>) { + chomp; + my @x = split(/\s+/); + + my $file = pop @x; + + my $cmd = "$FindBin::RealBin/../../../Trinity --single \"$file\" --output \"$file.trinity.$out_token\" $trin_args "; + + print "$cmd\n"; +} + +exit(0); + + + + + diff --git a/99.scripts/trinity_utils/util/misc/blast_outfmt6_group_segments.pl b/99.scripts/trinity_utils/util/misc/blast_outfmt6_group_segments.pl new file mode 100644 index 0000000..a51b131 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/blast_outfmt6_group_segments.pl @@ -0,0 +1,189 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib "$FindBin::RealBin/../../PerlLib"; +use Fasta_reader; +use List::Util qw(min max); +use Overlap_piler; +use Data::Dumper; + +my $usage = "\n\n\tusage: $0 blast.outfmt6 query_fasta target_fasta\n\n\n"; + +my $blast_file = $ARGV[0] or die $usage; +my $query_fasta = $ARGV[1] or die $usage; +my $target_fasta = $ARGV[2] or die $usage; + + +my %query_seq_lens = &get_seq_lengths($query_fasta); + +my %target_seq_lens; +if ($query_fasta eq $target_fasta) { + %target_seq_lens = %query_seq_lens; +} +else { + %target_seq_lens = &get_seq_lengths($target_fasta); +} + + + +my $MAX_MISSING = 10; +my $COUNT_MISSING = 0; + +main: { + + # header + print join("\t", "#query_acc", "target_acc", "avg_per_id", "min_Evalue", + "query_match_range", "target_match_range", "pct_query_len", "pct_target_len", "max_pct_len") . "\n"; + + + my @hits; + my $prev_query_target_pair = ""; + + open (my $fh, $blast_file) or die "Error, cannot open file $blast_file"; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my $query_acc = $x[0]; + my $target_acc = $x[1]; + + my $query_target_pair = join("$;", $query_acc, $target_acc); + if ($query_target_pair ne $prev_query_target_pair) { + &process_hits(@hits) if @hits; + @hits = (); + } + push (@hits, [@x]); + $prev_query_target_pair = $query_target_pair; + } + + # get last one + &process_hits(@hits); + + exit(0); +} + +#### +sub process_hits { + my @hits = @_; + + #print Dumper(\@hits); + + my @query_coords; + my @target_coords; + + my $query_acc = ""; + my $target_acc = ""; + + my $sum_pct_id_len = 0; + my $sum_len = 0; + + my $min_evalue = 1; + + foreach my $hit (@hits) { + my @fields = @$hit; + + unless ($query_acc) { + $query_acc = $fields[0]; + $target_acc = $fields[1]; + } + + my ($query_lend, $query_rend) = sort {$a<=>$b} ($fields[6], $fields[7]); + my ($target_lend, $target_rend) = sort {$a<=>$b} ($fields[8], $fields[9]); + + my $per_id = $fields[2]; + my $query_seg_len = $query_rend - $query_lend + 1; + $sum_pct_id_len += $query_seg_len * $per_id; + $sum_len += $query_seg_len; + + my $evalue = $fields[10]; + if ($evalue < $min_evalue) { + $min_evalue = $evalue; + } + + push (@query_coords, [$query_lend, $query_rend]); + push (@target_coords, [$target_lend, $target_rend]); + + } + + my $avg_per_id = sprintf("%.2f", $sum_pct_id_len / $sum_len); + + + my $query_match_len = 0; + my @query_match_regions; + { + + my @query_piles = &Overlap_piler::simple_coordsets_collapser(@query_coords); + foreach my $pile (@query_piles) { + my ($pile_lend, $pile_rend) = @$pile; + $query_match_len += $pile_rend - $pile_lend + 1; + push (@query_match_regions, "$pile_lend-$pile_rend"); + } + } + + my $target_match_len = 0; + my @target_match_regions; + { + my @target_piles = &Overlap_piler::simple_coordsets_collapser(@target_coords); + foreach my $pile (@target_piles) { + my ($pile_lend, $pile_rend) = @$pile; + $target_match_len += $pile_rend - $pile_lend + 1; + push (@target_match_regions, "$pile_lend-$pile_rend"); + } + } + + my $query_len = $query_seq_lens{$query_acc}; + my $target_len = $target_seq_lens{$target_acc}; + + if ( ! defined ($query_len)) { + print STDERR "Error, missing length for query: [$query_acc]\n"; + $COUNT_MISSING++; + if ($COUNT_MISSING > $MAX_MISSING) { + die "Error, too many missing seq length entries encountered\n"; + } + return; + } + + if (! defined($target_len)) { + print STDERR "Error, missing length for db hit: [$target_acc]\n"; + $COUNT_MISSING++; + if ($COUNT_MISSING > $MAX_MISSING) { + die "Error, too many missing seq length entries encountered\n"; + } + return; + } + + + my $pct_query_len = sprintf("%.2f", $query_match_len / $query_len * 100); + my $pct_target_len = sprintf("%.2f", $target_match_len / $target_len * 100); + + + print join("\t", $query_acc, $target_acc, $avg_per_id, $min_evalue, + join(",", @query_match_regions), join(",", @target_match_regions), + $pct_query_len, $pct_target_len, max($pct_query_len, $pct_target_len) ) . "\n"; + + + return; +} + + +#### +sub get_seq_lengths { + my ($fasta_file) = @_; + + my %seq_lens; + + my $fasta_reader = new Fasta_reader($fasta_file); + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + my $seq_len = length($seq_obj->get_sequence()); + + # print STDERR "$acc => $seq_len\n"; + + $seq_lens{$acc} = $seq_len; + } + + return(%seq_lens); +} + diff --git a/99.scripts/trinity_utils/util/misc/blast_outfmt6_group_segments.to_Markov_Clustering.pl b/99.scripts/trinity_utils/util/misc/blast_outfmt6_group_segments.to_Markov_Clustering.pl new file mode 100644 index 0000000..de58107 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/blast_outfmt6_group_segments.to_Markov_Clustering.pl @@ -0,0 +1,167 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Carp; +use Getopt::Long qw(:config posix_default no_ignore_case bundling pass_through); +use FindBin; +use lib "$FindBin::RealBin/../../PerlLib"; +use Pipeliner; +use File::Basename; + + +my $usage = <<__EOUSAGE__; + +####################################################################################### +# +# --outfmt6_grouped outfmt6 grouped output +# +# --min_pct_len minimum percent length covered by pairwise matches +# +# --min_per_id minimum percent identity +# +# --inflation_factor inflation factor for MCL clustering +# +####################################################################################### + + +__EOUSAGE__ + + ; + + + +my $help_flag; +my $outfmt6_grouped_file; +my $inflation_factor; +my $min_pct_len; +my $min_per_id; + +&GetOptions ( 'h' => \$help_flag, + + 'outfmt6_grouped=s' => \$outfmt6_grouped_file, + + 'min_pct_len=i' => \$min_pct_len, + + 'min_per_id=i' => \$min_per_id, + + 'inflation_factor=f' => \$inflation_factor, + + ); + +if ($help_flag) { + die $usage; +} + +if (@ARGV) { + die "Error, dont understand parameters @ARGV"; +} + + +unless ($outfmt6_grouped_file && $inflation_factor && $min_pct_len && $min_per_id) { + die $usage; +} + + +# add MCL to PATH setting +$ENV{PATH} = "/seq/regev_genome_portal/SOFTWARE/MCL/bin/:$ENV{PATH}"; + + + +main: { + + + my $filtered_hits = basename($outfmt6_grouped_file . ".minLEN_${min_pct_len}_pct_len.minPID_${min_per_id}.abc"); + my $checkpoint = ".$filtered_hits.ok"; + if (! -e $checkpoint) { + + my %best_hits; + + + + open (my $fh, $outfmt6_grouped_file) or die "Error, cannot open file $outfmt6_grouped_file"; + while (<$fh>) { + if (/^\#/) { next; } + chomp; + my @x = split(/\t/); + my ($transA, $transB, $per_id, $E_value, @rest) = split(/\t/); + my $per_len_match = pop @rest; + + if ($per_len_match >= $min_pct_len && $per_id >= $min_per_id) { + my $geneA = &parse_gene_name($transA); + my $geneB = &parse_gene_name($transB); + + if ($geneA eq $geneB) { next; } + + ($geneA, $geneB) = sort ($geneA, $geneB); + my $gene_pair_token = join("$;", $geneA, $geneB); + + + my $lowest_evalue = $best_hits{$gene_pair_token}; + if ( (! defined $lowest_evalue) || $lowest_evalue > $E_value) { + $best_hits{$gene_pair_token} = $E_value; + } + + + } + } + + # write best hits file + open (my $ofh, ">$filtered_hits") or die "Error, cannot write to $filtered_hits"; + foreach my $gene_pair_token (keys %best_hits) { + my ($geneA, $geneB) = split(/$;/, $gene_pair_token); + my $E_value = $best_hits{$gene_pair_token}; + print $ofh join("\t", $geneA, $geneB, $E_value) . "\n"; + } + close $ofh; + + `touch $checkpoint`; + } + + my $pipeliner = new Pipeliner(-verbose => 1); + my $cmd = "mcxload -abc $filtered_hits --stream-mirror --stream-neg-log10 " + . " -stream-tf 'ceil(200)' -o $filtered_hits.mci -write-tab $filtered_hits.tab"; + + $pipeliner->add_commands( new Command($cmd, ".$filtered_hits.tab.ok") ); + + $inflation_factor = sprintf("%.1f", $inflation_factor); + my $inflation_factor_dec_removed = $inflation_factor; + $inflation_factor_dec_removed =~ s/\.//; + + $cmd = "mcl $filtered_hits.mci -I $inflation_factor"; + my $mcl_outfile = "out.$filtered_hits.mci.I$inflation_factor_dec_removed"; + $pipeliner->add_commands( new Command($cmd, ".$mcl_outfile.ok")); + + + $cmd = "mcxdump -icl $mcl_outfile -tabr $filtered_hits.tab -o dump.$mcl_outfile"; + $pipeliner->add_commands( new Command($cmd, ".dump.$mcl_outfile.ok")); + + + $pipeliner->run(); + + + exit(0); + +} + +#### +sub parse_gene_name { + my ($trans_info) = @_; + + my ($gene_symbol, $trans_id); + if (/;/) { + ($trans_id, $gene_symbol) = split(/;/, $trans_info); + } + elsif (/\|/) { + ($gene_symbol, $trans_id) = split(/\|/, $trans_info); + } + + + if ($gene_symbol) { + return($gene_symbol); + } + else { + return($trans_info); + } +} diff --git a/99.scripts/trinity_utils/util/misc/blast_outfmt6_group_segments.tophit_coverage.pl b/99.scripts/trinity_utils/util/misc/blast_outfmt6_group_segments.tophit_coverage.pl new file mode 100644 index 0000000..90f3d32 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/blast_outfmt6_group_segments.tophit_coverage.pl @@ -0,0 +1,110 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 blast.grouped\n\n"; + +my $blast_out = $ARGV[0] or die $usage; + +main: { + + + my $counter = 0; + + my %query_to_top_hit; # only storing the hit with the greatest blast score. + + # outfmt6: + # qseqid sseqid pident length mismatch gapopen qstart qend sstart send evalue bitscore + + open (my $fh, $blast_out) or die "Error, cannot open file $blast_out"; + while (<$fh>) { + if (/^\#/) { next; } + chomp; + my $line = $_; + my @x = split(/\t/); + my $query_id = $x[0]; + my $db_id = $x[1]; + my $percent_id = $x[2]; + my $Evalue = $x[3]; + + my $pct_target_len = $x[7]; + + + if ( (! exists $query_to_top_hit{$query_id}) || ($Evalue < $query_to_top_hit{$query_id}->{Evalue}) + || + ($Evalue == $query_to_top_hit{$query_id}->{Evalue} + && + $pct_target_len > $query_to_top_hit{$query_id}->{pct_target_len} ) + ) { + + $query_to_top_hit{$query_id} = { query_id => $query_id, + db_id => $db_id, + percent_id => $percent_id, + Evalue => $Evalue, + pct_target_len => $pct_target_len, + }; + + } + } + close $fh; + + + ## get the best transcript hit per db ID + + my %db_id_to_trans_hits; + + foreach my $hit_struct (values %query_to_top_hit) { + + my $db_id = $hit_struct->{db_id}; + + push (@{$db_id_to_trans_hits{$db_id}}, $hit_struct); + } + + + ## histogram summary + + my @bins = qw(10 20 30 40 50 60 70 80 90 100); + my %bin_counts; + + + foreach my $db_id (keys %db_id_to_trans_hits) { + + my @hit_structs = @{$db_id_to_trans_hits{$db_id}}; + + @hit_structs = sort {$a->{Evalue} <=> $b->{Evalue} + || + $b->{pct_target_len} <=> $a->{pct_target_len} } @hit_structs; + + + my $entry = shift @hit_structs; # take the lowest E-value w/ longest pct_target_len + + my $pct_cov = $entry->{pct_target_len}; + + my $prev_bin = 0; + foreach my $bin (@bins) { + if ($pct_cov > $prev_bin && $pct_cov <= $bin) { + $bin_counts{$bin}++; + } + $prev_bin = $bin; + } + + } + + + ## Report counts per bin + print "#hit_pct_cov_bin\tcount_in_bin\t>bin_below\n"; + + my $cumul = 0; + foreach my $bin (reverse(@bins)) { + my $count = $bin_counts{$bin} || 0; + $cumul += $count; + print join("\t", $bin, $count, $cumul) . "\n"; + } + + + + exit(0); + + +} diff --git a/99.scripts/trinity_utils/util/misc/blastn_wrapper.pl b/99.scripts/trinity_utils/util/misc/blastn_wrapper.pl new file mode 100644 index 0000000..5215bdb --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/blastn_wrapper.pl @@ -0,0 +1,41 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 db query [opts]\n\n"; + +unless (@ARGV) { + die $usage; +} +my $db = $ARGV[0] or die $usage; +my $query = $ARGV[1] or die $usage; + +shift @ARGV; +shift @ARGV; + +main: { + + my $cmd = "makeblastdb -in $db -dbtype nucl"; + &process_cmd($cmd) unless (-s "$db.nin"); # only build it once + + $cmd = "blastn -db $db -query $query -dust no @ARGV"; + &process_cmd($cmd); + + exit(0); +} + +#### +sub process_cmd { + my ($cmd) = @_; + + my $ret = system($cmd); + if ($ret) { + die "Error, CMD: $cmd died with ret $ret"; + } + + return; +} + + + diff --git a/99.scripts/trinity_utils/util/misc/blat_util/blat_sam_add_reads2.pl b/99.scripts/trinity_utils/util/misc/blat_util/blat_sam_add_reads2.pl new file mode 100644 index 0000000..4177bf8 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/blat_util/blat_sam_add_reads2.pl @@ -0,0 +1,85 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../../PerlLib"); +use Nuc_translator; + +my $usage = "usage: $0 blat.psl.nameSorted.sam reads.tab.nameSorted\n\n"; + + +my $sam = $ARGV[0] or die $usage; +my $seqs = $ARGV[1] or die $usage; + + +main: { + + open (my $sam_fh, "$sam") or die "Error, cannot open file $sam"; + open (my $seqs_fh, "$seqs") or die "Error, cannot open file $seqs"; + + my $seq_line = <$seqs_fh>; + chomp $seq_line; + my ($seq_acc, $seq, $qual) = split(/\t/, $seq_line); + $seq_acc =~ s/\s//g; # rid any ws from acc name + + unless ($qual) { + $qual = 'B' x length($seq); + } + + while (my $sam_line = <$sam_fh>) { + + my @x = split(/\t/, $sam_line); + + my $acc = $x[0]; + my $flag = $x[1]; + + my $aligned_orient = ($flag & 0x0010) ? '-' : '+'; + + + while ($seq_acc lt $acc) { + $seq_line = <$seqs_fh>; + chomp $seq_line; + ($seq_acc, $seq, $qual) = split(/\t/, $seq_line); + $seq_acc =~ s/\s//g; # no ws in acc name + unless (defined $qual) { + $qual = 'B' x length($seq); #$qual = "*"; + } + } + + if ($acc eq $seq_acc) { + + if ($aligned_orient eq '-') { + my $revseq = &reverse_complement($seq); + my @q = split(//, $qual); + my $revqual = join("", reverse(@q)); + $x[9] = $revseq; + $x[10] = $revqual; + } + else { + $x[9] = $seq; + $x[10] = $qual; + } + } + else { + die "Error,\n[$acc]\nnot encountered in file: $seqs,\ncurrently cursor is at seq:\n[$seq_acc]\n"; + } + + ## set read quality score to a high value so that Scripture will use it. + $x[4] = 255; + $x[5] =~ s/H/S/gi; # convert hard to soft clips; samtools doesn't like H's + + foreach my $val (@x) { + unless (defined $val) { + $val = "*"; + } + } + + print join("\t", @x); + } + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/blat_util/blat_to_sam.pl b/99.scripts/trinity_utils/util/misc/blat_util/blat_to_sam.pl new file mode 100644 index 0000000..5b9be58 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/blat_util/blat_to_sam.pl @@ -0,0 +1,125 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use Cwd; + +use Getopt::Long qw(:config no_ignore_case bundling); + + +$ENV{LC_ALL} = 'C'; + +my $usage = <<_EOUSAGE_; + +################################################################################################ +# +# Required: +# +# --genome genome in fasta format +# --reads reads in fasta format +# +# Optional: +# +# --blat_params quote-delimited params to pass to blat, eg. "-q=rna -t=dna -maxIntron=1000" (default: "-q=rna -t=dna") +# +# --top number of top hits (default: 10) +# --min_per_ID minimum percent identity (default: 95) +# +############################################################################################## + + +_EOUSAGE_ + + ; + + +my ($genome_fa, $reads_fa); + +my $blat_params = "-q=rna -t=dna"; +my $top_hits = 10; +my $min_per_ID = 95; + + + +&GetOptions( 'genome=s' => \$genome_fa, + 'reads=s' => \$reads_fa, + + 'blat_params=s' => \$blat_params, + + 'top=i' => \$top_hits, + 'min_per_ID=i' => \$min_per_ID, + ); + +unless ($genome_fa && $reads_fa) { + die $usage; +} + + +{ + my @required_progs = qw (blat psl2sam.pl); + + foreach my $prog (@required_progs) { + my $path = `sh -c "command -v $prog"`; + unless ($path =~ /^\//) { + die "Error, cannot locate required program: $prog"; + } + } +} + + + +main: { + + my $util_dir = "$FindBin::RealBin/../util"; + + my $cmd = "$util_dir/fasta_to_tab.pl $reads_fa > $reads_fa.tab"; + &process_cmd($cmd) unless (-s "$reads_fa.tab"); + + $cmd = "sort -T . -S 2G -k 1,1 $reads_fa.tab > $reads_fa.tab.sort"; + &process_cmd($cmd) unless (-s "$reads_fa.tab.sort"); + + # run blat + $cmd = "blat $blat_params $genome_fa $reads_fa $reads_fa.psl"; + &process_cmd($cmd); + + # convert to sam + $cmd = "psl2sam.pl -q 0 -r 0 $reads_fa.psl | sort -T . -S 2G -k 1,1 > $reads_fa.psl.sam"; + &process_cmd($cmd); + + + # add the reads + $cmd = "$util_dir/blat_sam_add_reads2.pl $reads_fa.psl.sam $reads_fa.tab.sort > $reads_fa.psl.sam.wReads"; + &process_cmd($cmd); + + ## prune output to top matches: + $cmd = "$util_dir/top_blat_sam_extractor.pl $reads_fa.psl.sam.wReads $top_hits $min_per_ID > $reads_fa.psl.sam.wReads.top"; + &process_cmd($cmd); + + $cmd = "$FindBin::RealBin/cigar_tweaker $reads_fa.psl.sam.wReads.top $genome_fa > $reads_fa.psl.sam.wReads.top.tweaked"; + &process_cmd($cmd); + + $cmd = "sort -T . -S 2G -k 3,3 -k 4,4n $reads_fa.psl.sam.wReads.top.tweaked > $reads_fa.psl.sam.wReads.top.tweaked.coordSorted.sam"; + &process_cmd($cmd); + + + + exit(0); +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print "CMD: $cmd\n"; + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + diff --git a/99.scripts/trinity_utils/util/misc/blat_util/blat_top_hit_extractor.pl b/99.scripts/trinity_utils/util/misc/blat_util/blat_top_hit_extractor.pl new file mode 100644 index 0000000..733eed1 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/blat_util/blat_top_hit_extractor.pl @@ -0,0 +1,121 @@ +#!/usr/bin/env perl + +use strict; + +my $SEE = 0; + +############### +# blat format: # Q=cDNA T=genomic +################ + +# 0: match +# 1: mis-match +# 2: rep. match +# 3: N's +# 4: Q gap count +# 5: Q gap bases +# 6: T gap count +# 7: T gap bases +# 8: strand +# 9: Q name +# 10: Q size +# 11: Q start +# 12: Q end +# 13: T name +# 14: T size +# 15: T start +# 16: T end +# 17: block count +# 18: block Sizes +# 19: Q starts +# 20: T starts +# 21: Q seqs (pslx format) +# 22: T seqs (pslx format) + +############################################################# +## This script filters blat output and extracts only ## +## the top scoring alignment chain for each accession. ## +############################################################# + +my $line_num = 0; +my %data; + +my $filename = $ARGV[0] or die "\n\nusage: $0 outputfile.psl [num_top_hits=1]\n\n"; +my $num_top_hits = $ARGV[1] || 1; + +## First Pass, assign scores to entries associated with accessions. +open (FILE, "$filename"); +my $line_num = 0; +while () { + $line_num++; + my @x = split (/\t/); + unless ($x[0] =~ /^\d/) {next;} + my ($matches, $mismatches, $q_gap_num, $t_gap_num, $q_insert, $t_insert) = ($x[0], $x[1], $x[4], $x[6], $x[5], abs($x[7])); + + my $score = &calculate_score($matches, $mismatches, $q_gap_num, $t_gap_num, $q_insert, $t_insert); + + my $accession = $x[9]; + my $score_struct = {score=>$score, + line_num=>$line_num}; + push (@{$data{$accession}}, $score_struct); +} + +close FILE; + +## Identify each highest scoring match +my %line_nums_to_print; + +foreach my $accession (keys %data) { + my @hits = @{$data{$accession}}; + @hits = reverse sort {$a->{score}<=>$b->{score}} @hits; #sort in reverse order of score. + + if ($SEE) { #verify the top hit is chosen. + foreach my $hit (@hits) { + print $hit->{line_num} . ":" . $hit->{score} . " "; + } + + print "\n chose "; + } + + for (my $i = 0; $i < $num_top_hits && $i <= $#hits; $i++) { + + my $top_hit = $hits[$i]; + print $top_hit->{line_num} . ":" . $top_hit->{score} . "\n" if $SEE; + my $line_num = $top_hit->{line_num}; + $line_nums_to_print{$line_num} = 1; + } +} + +open (FILE, "$filename"); +$line_num = 0; +while () { + $line_num++; + if ($line_nums_to_print{$line_num}) { + print; + } +} +close FILE; + +exit(0); + + +#### +sub calculate_score { + my ($match, $mismatch, $q_gap_num, $t_gap_num, $q_insert, $t_insert) = @_; + + ## JKent's code in pslFilter: + # score = (psl->match + psl->repMatch)*reward - psl->misMatch*cost + # - (psl->qNumInsert + psl->tNumInsert + 1) * gapOpenCost + # - log(psl->qBaseInsert + psl->tBaseInsert + 1) * gapSizeLogMod; + + #use jkent's default score parameters: + my $reward = 1; + my $cost = 1; + my $gapOpenCost = 4; + my $gapSizeLogMod = 1; + + return ( ($match * $reward) - ($mismatch * $cost) - + ( ($q_gap_num + $t_gap_num) * $gapOpenCost) - + ( log ($q_insert + $t_insert + 1) * $gapSizeLogMod) ); + +} diff --git a/99.scripts/trinity_utils/util/misc/blat_util/process_BLAT_alignments.pl b/99.scripts/trinity_utils/util/misc/blat_util/process_BLAT_alignments.pl new file mode 100644 index 0000000..6d63718 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/blat_util/process_BLAT_alignments.pl @@ -0,0 +1,277 @@ +#!/usr/bin/env perl + +use FindBin; +use lib ("$FindBin::RealBin/../../../PerlLib"); + +use strict; +use warnings; +use threads; + +use Fasta_reader; +use Process_cmd; +use Thread_helper; + +use Getopt::Long qw (:config no_ignore_case bundling); +use vars qw ($DEBUG $opt_h $opt_g $opt_t $opt_c $opt_o $opt_B $opt_I); + +my $CPU = 1; +my $num_top_hits = 1; +my $KEEP_PSLX = 0; + +&GetOptions( 'g=s' => \$opt_g, + 'd' => \$DEBUG, + 'h' => \$opt_h, + 'c=s' => \$opt_c, + 'o=s' => \$opt_o, + 't=s' => \$opt_t, + 'I=i' => \$opt_I, + 'N=i' => \$num_top_hits, + 'CPU=i' => \$CPU, + 'KEEP_PSLX' => \$KEEP_PSLX, + ); + +our $SEE = 0; + +$|++; + +my $MAX_INTRON = $opt_I || 100000; + +my $usage = <<_EOH_; + +Script chunks EST alignments into more manageable data sets. + +############################# Options ############################### +# +# -g genomic_seq.db +# -t transcripts database +# -I maximum intron length (default: 100000) +# -o prefix for output file (default: 'blat') +# -N number of top hits (default: $num_top_hits) +# +# --CPU number of threads (default: 1) +# +# -h this help menu +# -d debug mode +# --KEEP_PSLX retain the raw blat output files +# +###################### Process Args and Options ##################### + + +_EOH_ + + ; + + +my $genome_db = $opt_g; +my $transcript_db = $opt_t; +my $output_prefix = $opt_o || "blat"; +my $blat_path = "blat"; +my $util_dir = $FindBin::RealBin; + +unless ($genome_db && $transcript_db) { + die "$usage\n"; +} + +my $ooc_cmd = "$blat_path $genome_db $transcript_db -q=rna -dots=100 -maxIntron=$MAX_INTRON -makeOoc=11.ooc nada"; + +my $blat_thr; + +unless (-s "11.ooc") { + $blat_thr = threads->create('process_cmd', $ooc_cmd); +} + +my $blat_partitions_dir = "blat_out_dir"; +unless (-d $blat_partitions_dir) { + mkdir($blat_partitions_dir) or die "Error, cannot mkdir $blat_partitions_dir"; +} + +my $num_seqs = `grep '>' $transcript_db | wc -l `; +$num_seqs =~ s/\s//g; +unless ($num_seqs && $num_seqs =~ /^\d+$/) { + die "Error, cannot determine number of fasta entries in $transcript_db"; +} + +my $seqs_per_partition = int($num_seqs/$CPU + 0.5); +unless ($seqs_per_partition > 1) { + $seqs_per_partition = 1; +} + +my @transcript_files = &partition_transcript_db($transcript_db, $seqs_per_partition, $blat_partitions_dir); + +if ($blat_thr) { + $blat_thr->join(); + if ($blat_thr->error()) { + die "Error, $ooc_cmd died ..."; + } +} + + +############################### +## Run BLAT +############################### +my @pslx_files; + +{ + + my $thread_helper = new Thread_helper($CPU); + my %thread_to_checkpoint; + + foreach my $transcript_file (@transcript_files) { + + + ## process blat search: + my $cmd = "$blat_path $genome_db $transcript_file -q=rna -dots=100 " + . " -maxIntron=$MAX_INTRON -out=pslx -ooc=11.ooc $transcript_file.pslx"; + + + my $checkpoint_file = "$transcript_file.pslx.completed"; + unless (-e $checkpoint_file) { + + my $thread = threads->create('process_cmd', $cmd); + $thread_helper->add_thread($thread); + + my $thread_id = $thread->tid(); + $thread_to_checkpoint{$thread_id} = {thread => $thread, + checkpoint => $checkpoint_file, + }; + } + push (@pslx_files, "$transcript_file.pslx"); + + } + + $thread_helper->wait_for_all_threads_to_complete(); + + ## write checkpoints for successful threads: + foreach my $thread_id (keys %thread_to_checkpoint) { + my $thread_info_href = $thread_to_checkpoint{$thread_id}; + my $thread = $thread_info_href->{thread}; + unless ($thread->error()) { + my $checkpoint = $thread_info_href->{checkpoint}; + system("touch $checkpoint"); + } + } + + if (my @failures = $thread_helper->get_failed_threads()) { + die "Error, ". scalar(@failures) . " blat searches failed. "; + } +} + +###################### +## get top hits. +###################### + + +my @top_hits_files; + + +{ + + my %thread_to_checkpoint; + my $thread_helper = new Thread_helper($CPU); + + foreach my $pslx_file (@pslx_files) { + + my $cmd = "$util_dir/blat_top_hit_extractor.pl $pslx_file $num_top_hits > $pslx_file.top_${num_top_hits}"; + + my $completed_checkpoint_file = "$pslx_file.top_${num_top_hits}.completed"; + unless (-e $completed_checkpoint_file) { + + my $thread = threads->create('process_cmd', $cmd); + $thread_helper->add_thread($thread); + + my $thread_id = $thread->tid(); + $thread_to_checkpoint{$thread_id} = { thread => $thread, + checkpoint => $completed_checkpoint_file, + }; + + } + push (@top_hits_files, "$pslx_file.top_${num_top_hits}"); + } + + $thread_helper->wait_for_all_threads_to_complete(); + + ## write checkpoints for successful threads: + foreach my $thread_id (keys %thread_to_checkpoint) { + my $thread_info_href = $thread_to_checkpoint{$thread_id}; + my $thread = $thread_info_href->{thread}; + unless ($thread->error()) { + my $checkpoint = $thread_info_href->{checkpoint}; + system("touch $checkpoint"); + } + } + + + if (my @failures = $thread_helper->get_failed_threads()) { + die "Error, ". scalar(@failures) . " blat top hit selectors failed. "; + } +} + + + +if (-s "$output_prefix.gff3") { + print STDERR "WARNING, REPLACING EXISTING FILE: $output_prefix.gff3. KILL THIS NOW TO PREVENT THIS. (you have 10 seconds)\n"; + sleep(10); + print STDERR "OK, too late. replacing it now.\n"; + unlink("$output_prefix.gff3"); +} + +foreach my $top_hits_file (@top_hits_files) { + + # convert to gff3 format + print STDERR "-converting $top_hits_file to gff3\n"; + + my $cmd = "$util_dir/pslx_to_gff3.pl < $top_hits_file >> $output_prefix.gff3"; + &process_cmd($cmd); +} + +## clean up the pslx files we no longer need. +foreach my $pslx_file (@pslx_files) { + unlink($pslx_file) unless $KEEP_PSLX; # these files can be huge. Once have top hits, no longer need all hits (hopefully). +} + +print STDERR "done.\n"; + +exit(0); + + +#### +sub partition_transcript_db { + my ($transcript_db, $seqs_per_partition, $blat_partitions_dir) = @_; + + my @files; + + my $checkpoint_file = "$blat_partitions_dir/partitions.completed"; + + if (-e $checkpoint_file) { + my @files = <$blat_partitions_dir/partition.*.fa>; + return(@files); + } + else { + + + my $fasta_reader = new Fasta_reader($transcript_db); + my $partition_counter = 0; + + my $counter = 0; + my $ofh; + + while (my $seq_obj = $fasta_reader->next()) { + my $fasta_entry = $seq_obj->get_FASTA_format(); + if ($counter % $seqs_per_partition == 0) { + close $ofh if $ofh; + $partition_counter++; + my $outfile = "$blat_partitions_dir/partition.$counter.fa"; + open ($ofh, ">$outfile") or die "Error, cannot write to outfile: $outfile"; + push (@files, $outfile); + } + print $ofh $fasta_entry; + $counter++; + } + + close $ofh if $ofh; + + system("touch $checkpoint_file"); + + return(@files); + } +} diff --git a/99.scripts/trinity_utils/util/misc/blat_util/pslx_to_gff3.pl b/99.scripts/trinity_utils/util/misc/blat_util/pslx_to_gff3.pl new file mode 100644 index 0000000..0451cbb --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/blat_util/pslx_to_gff3.pl @@ -0,0 +1,242 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + + +################ +# blat format: # Q=cDNA T=genomic +################ + +# 0: match +# 1: mis-match +# 2: rep. match +# 3: N's +# 4: Q gap count +# 5: Q gap bases +# 6: T gap count +# 7: T gap bases +# 8: strand +# 9: Q name +# 10: Q size +# 11: Q start +# 12: Q end +# 13: T name +# 14: T size +# 15: T start +# 16: T end +# 17: block count +# 18: block Sizes +# 19: Q starts +# 20: T starts +# 21: Q seqs (pslx format) +# 22: T seqs (pslx format) + +## All sequences start at 0 here; array-based. + +my $JOIN_GAP = 9; #join alignment segments if within this gap along the genomic sequence. + +my $chain_number = 0; + +while () { + unless (/\w/) { next; } + chomp; + my @x = split (/\t/); + unless ($x[0] =~ /^\d/) {next;} #eliminate headers if present. + my @alignment_segments; + $chain_number++; + my $strand = $x[8]; + my $cdna_name = $x[9]; + my $genomic_name = $x[13]; + my $genomic_length = $x[14]; + my $cdna_length = $x[10]; + my $cDNA_seqs = $x[21]; + my $genomic_seqs = $x[22]; + my $num_segs = $x[17]; + + my @cdna_coords = split (/,/, $x[19]); + my @genomic_coords = split (/,/, $x[20]); + my @lengths = split (/,/, $x[18]); + my @cDNA_seqs = split (/,/, $x[21]); + my @genomic_seqs = split (/,/, $x[22]); + + ## going to implement score as chain_score + segment score + ## Chain score = matches_num - mismatch_num + my $chain_score = $x[0] - $x[1]; + + ## report each segment match as a separate btab entry: + my $segment_number = 0; + for (my $i = 0; $i < $num_segs; $i++) { + $segment_number++; + my $length = $lengths[$i]; + my $cdna_coord = $cdna_coords[$i]; + my $genomic_coord = $genomic_coords[$i]; + my ($cdna_end5, $cdna_end3) = (++$cdna_coord, $cdna_coord + $length - 1); + my ($genomic_end5, $genomic_end3) = (++$genomic_coord, $genomic_coord + $length -1); + if ($strand eq "-") { + ($genomic_end5, $genomic_end3) = ($genomic_end3, $genomic_end5); + ($cdna_end5, $cdna_end3) = sort {$a<=>$b} ($cdna_length - $cdna_end5 + 1, $cdna_length - $cdna_end3 + 1); + } + my $segment_score = $chain_score + $length; + my $per_id = -1; + my ($gseq, $cseq); + if ( ($gseq = $genomic_seqs[$i]) && ($cseq = $cDNA_seqs[$i])) { + if ($gseq eq $cseq) { + $per_id = 100; + } else { + ## walk thru and determine num ids + my $num_id = 0; + my @gseq_array = split (//, $gseq); + my @cseq_array = split (//, $cseq); + for (my $j = 0; $j < $length; $j++) { + if ($gseq_array[$j] eq $cseq_array[$j]) { + $num_id++; + } + } + $per_id = ($num_id/$length) * 100; + } + } + + ## Create btab line. + my @btab; + $btab[0] = $genomic_name; + $btab[2] = $genomic_length; + $btab[3] = "blat"; + $btab[5] = $cdna_name; + $btab[6] = $genomic_end5; + $btab[7] = $genomic_end3; + $btab[8] = $cdna_end5; + $btab[9] = $cdna_end3; + + $btab[10] = $per_id; + $btab[12] = $segment_score; + + $btab[13] = $chain_number; + $btab[14] = $segment_number; + + $btab[18] = $length; + push (@alignment_segments, [@btab]); + + } + + &process_alignment_chain(\@alignment_segments, $chain_number); +} + +exit(0); + + +## Join alignment segments if within 5 bp along the genomic sequence +sub process_alignment_chain { + my $alignment_segments_aref = shift; + my $chain_number = shift; + my @segments = sort {$a->[8]<=>$b->[8]} @$alignment_segments_aref; + if ($#segments > 0) { + my @new_segments = ($segments[0]); # always holds the last segment analyzed. + for (my $i=1; $i <= $#segments; $i++) { + my $last_segment = $new_segments[$#new_segments]; + my $current_segment = $segments[$i]; + my $prev_end3 = $last_segment->[7]; + my $curr_end5 = $current_segment->[6]; + my $gap_length = abs ($prev_end3 - $curr_end5) - 1; + + if ($gap_length <= $JOIN_GAP) { + ## Must join prev and current segment + my $prev_seg_length = abs ($last_segment->[7] - $last_segment->[6]) + 1; + my $curr_seg_length = abs ($current_segment->[7] - $current_segment->[6]) + 1; + + ## make prev end3 the new end3 for both genomic and cdna coordinates + $last_segment->[7] = $current_segment->[7]; + $last_segment->[9] = $current_segment->[9]; + + ## recalculate the percent ID + my $prev_per_id = $last_segment->[10]; + my $curr_per_id = $current_segment->[10]; + + my $new_per_id = ($prev_seg_length * $prev_per_id + $curr_seg_length * $curr_per_id) / + ($prev_seg_length + $curr_seg_length + $gap_length); + + $last_segment->[10] = $new_per_id; + $last_segment->[18] = abs($last_segment->[9] - $last_segment->[8]) + 1; + $last_segment->[12] += $curr_seg_length; + + } else { + + #make the current segment the last segment + push (@new_segments, $current_segment); + + } + } + @segments = @new_segments; + + ## renumber the segment numbers + my $segnum = 0; + foreach my $segment (@segments) { + $segnum++; + $segment->[14] = $segnum; + } + } + + ## write GFF format. + my $orient; + foreach my $segment (@segments) { + + my $genomic_end5 = $segment->[6]; + my $genomic_end3 = $segment->[7]; + + unless ($orient) { + if ($genomic_end5 < $genomic_end3) { + $orient = '+'; + } + elsif ($genomic_end3 < $genomic_end5) { + $orient = '-'; + } + } + if ($orient) { + last; + } + } + + unless ($orient) { + $orient = '+'; ## set a default + } + + + foreach my $segment (@segments) { + + my $genomic_contig = $segment->[0]; + my $cdna_name = $segment->[5]; + + my $genomic_end5 = $segment->[6]; + my $genomic_end3 = $segment->[7]; + + my ($genomic_lend, $genomic_rend) = sort {$a<=>$b} ($genomic_end3, $genomic_end5); + + my $cdna_end5 = $segment->[8]; + my $cdna_end3 = $segment->[9]; + + my $per_id = sprintf("%.2f", $segment->[10]); + + my $chain_ID = "blat.proc$$.chain_" . $chain_number; + + print join("\t", $genomic_contig, "BLAT", "cDNA_match", + $genomic_lend, $genomic_rend, $per_id, $orient, ".", + "ID=$chain_ID;Target=$cdna_name $cdna_end5 $cdna_end3 +") . "\n"; + + + + + } + + return; +} + + + + + + + + + + + diff --git a/99.scripts/trinity_utils/util/misc/blat_util/run_BLAT_shortReads.pl b/99.scripts/trinity_utils/util/misc/blat_util/run_BLAT_shortReads.pl new file mode 100644 index 0000000..d2b2c9b --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/blat_util/run_BLAT_shortReads.pl @@ -0,0 +1,36 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 genome_db transcript_db maxIntron outFile\n\n"; + +my $genome_db = $ARGV[0] or die $usage; +my $transcript_db = $ARGV[1] or die $usage; +my $max_intron = $ARGV[2] or die $usage; +my $outFile = $ARGV[3] or die $usage; + +main: { + + my $ooc_cmd = "blat -t=dna -q=rna -maxIntron=$max_intron -makeOoc=11.ooc $genome_db $transcript_db $outFile"; + unless (-s "11.ooc") { + &process_cmd($ooc_cmd); + } + + + my $blat_cmd = "blat -t=dna -q=rna -maxIntron=$max_intron -ooc=11.ooc $genome_db $transcript_db $outFile"; + &process_cmd($blat_cmd); + + exit(0); +} + +sub process_cmd { + my ($cmd) = @_; + + my $ret = system($cmd); + if ($ret) { + die "Error, $cmd died with ret $ret"; + } + + return; +} diff --git a/99.scripts/trinity_utils/util/misc/blat_util/top_blat_sam_extractor.pl b/99.scripts/trinity_utils/util/misc/blat_util/top_blat_sam_extractor.pl new file mode 100644 index 0000000..d8842d5 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/blat_util/top_blat_sam_extractor.pl @@ -0,0 +1,132 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../../PerlLib"); + +use SAM_reader; +use SAM_entry; + + +my $usage = "usage: $0 blat.nameSorted.sam [num_top_hits=20] [min_per_ID=0]\n\n"; + +# note min perID is based on read length and blat2sam.pl default scoring with -q 0 -r 0 + +my $blat_sam = $ARGV[0] or die $usage; +my $num_top_hits = $ARGV[1] || 20; +my $min_per_ID = $ARGV[2] || 0; + +main: { + + my $sam_reader = new SAM_reader($blat_sam); + + while ($sam_reader->has_next()) { + + my @entries; + + my $sam_entry = $sam_reader->get_next(); + push (@entries, $sam_entry); + + while ($sam_reader->has_next() + && + $sam_reader->preview_next()->get_read_name() eq $sam_entry->get_read_name()) { + + push (@entries, $sam_reader->get_next()); + } + + &report_top_hits(@entries); + } + + exit(0); + +} + + +#### +sub report_top_hits { + my @entries = @_; + + @entries = &get_top_hits(@entries); + + + foreach my $entry (@entries) { + print $entry->toString() . "\n"; + } + + return; +} + + + +#### +sub get_top_hits { + my @entries = @_; + + my @structs; + + foreach my $entry (@entries) { + my @fields = $entry->get_fields(); + + my $seq_length = length($entry->get_sequence()); + + my $min_score = $seq_length - ( ( 1 - ($min_per_ID / 100)) * $seq_length * 3); # perfect_score - (num_mismatches * mismatch_penalty) + + my $alignment_field = $fields[11]; + + $alignment_field =~ /AS:i:(\d+)/ or die "Error, no score reported for " . $entry->toString(); + + my $score = $1; + unless (defined $score) { + die "Error, no score for " . $entry->toString(); + } + + #print "MIN_score: $min_score vs. score: $score\n"; + + if ($score >= $min_score) { + + push (@structs, { entry => $entry, + score => $score, + } + ); + + } + } + + @structs = reverse sort {$a->{score}<=>$b->{score}} @structs; + + + + if ($num_top_hits < 0 && scalar(@structs) > 1) { + # multiply mapped reads + # ignoring entry. + return(); + } + + elsif ($num_top_hits > 0 && scalar (@structs) > $num_top_hits) { + @structs = @structs[0..$num_top_hits-1]; + } + + ## unwrap: + my @ret; + if (@structs) { + my $top_struct = shift @structs; + + my $top_score = $top_struct->{score}; + + push (@ret, $top_struct->{entry}); + + foreach my $struct (@structs) { + if ($struct->{score} == $top_score) { + push (@ret, $struct->{entry}); + } + else { + last; + } + } + } + + return(@ret); +} + diff --git a/99.scripts/trinity_utils/util/misc/capture_orig_n_unmapped_reads.pl b/99.scripts/trinity_utils/util/misc/capture_orig_n_unmapped_reads.pl new file mode 100644 index 0000000..b923f15 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/capture_orig_n_unmapped_reads.pl @@ -0,0 +1,224 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use Cwd; +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + +use Fastq_reader; +use SAM_reader; +use SAM_entry; + +my $usage = <<_EOUSAGE_; + + +############################################################################################################ +# +# --aligned_sam aligned reads in sam format (sorting doesn't matter) +# +# --sampled_fq_list comma-delimited list of sampled files (no spaces) +# eg. "sample_left.fq,sample_right.fq" or "ssample_single.fq" +# +# --all_fq_list same as above, corresponding to the original (complete) set of fastq reads +# +############################################################################################################# + + +_EOUSAGE_ + + + ; + + +my $aligned_sam_file; +my $sampled_fq_list; +my $all_fq_list; + + +&GetOptions( 'aligned_sam=s' => \$aligned_sam_file, + 'sampled_fq_list=s' => \$sampled_fq_list, + 'all_fq_list=s' => \$all_fq_list, + ); + + +if (@ARGV) { + die $usage; +} + +unless ($aligned_sam_file && $sampled_fq_list && $all_fq_list) { + die $usage; +} + + + +main: { + + ## get the list of reads that were in the original sample + my %sampled_reads; + foreach my $fq_file (split(/,/, $sampled_fq_list)) { + $fq_file =~ s/\s+//g; + + &get_sampled_read_names(\%sampled_reads, $fq_file); + } + + my $num_sampled_reads = scalar(keys %sampled_reads); + print STDERR "$num_sampled_reads number of sampled reads\n"; + + + ## identify those reads that are aligned + my %aligned_reads; + &parse_aligned_reads(\%aligned_reads, $aligned_sam_file); + + my $num_aligned_reads = scalar(keys %aligned_reads); + print STDERR "$num_aligned_reads number of aligned reads\n"; + + + ## report the unmapped reads and the original ones. + foreach my $fq_file (split(/,/, $all_fq_list)) { + $fq_file =~ s/\s//g; + + &report_orig_n_unmapped(\%sampled_reads, \%aligned_reads, $fq_file); + } + + + exit(0); + +} + + +#### +sub get_sampled_read_names { + my ($sampled_reads_href, $fq_file) = @_; + + my $x = 0; + my $fq_parser = new Fastq_reader($fq_file); + while (my $record = $fq_parser->next()) { + my $full_read_name = $record->get_full_read_name(); + #print "SAMPLED: [$full_read_name]\n"; + $sampled_reads_href->{$full_read_name}++; + + $x++; + #if ($x > 5) { last; } + } + + return; +} + + +#### +sub parse_aligned_reads { + my ($aligned_reads_href, $aligned_sam_file) = @_; + + my $sam_reader = new SAM_reader($aligned_sam_file); + + my $num_reads_aligned = 0; + + while (my $sam_entry = $sam_reader->get_next()) { + + if ($sam_entry->is_query_unmapped()) { next; } + + my $read_name = $sam_entry->reconstruct_full_read_name(); + + #print "SAM: $read_name\n"; + + $aligned_reads_href->{$read_name}++; + + $num_reads_aligned++; + + + #if ($num_reads_aligned > 5) { last; } + } + + #print "$aligned_sam_file contains $num_reads_aligned aligned reads\n\n"; + + return; +} + +#### +sub report_orig_n_unmapped { + my ($sampled_reads_href, $aligned_reads_href, $fq_file) = @_; + + my $fq_parser = new Fastq_reader($fq_file); + + my $num_reads_output = 0; + my $num_reads_skipped = 0; + + + + while (my $record = $fq_parser->next()) { + + my $read_name = $record->get_full_read_name(); + + #print "FQ: [$read_name]\n"; + + my $print_record_flag = $sampled_reads_href->{$read_name} || 0; + + + #print "sampled: $read_name = $print_record_flag\n"; + + #$print_record_flag = 0; + + unless ($print_record_flag) { + + if ($read_name =~ /^(\S+)\/([12])$/) { + + ## check to see if both read pairs align. If not, report both. + + my $core = $1; + my $pair_val = $2; + + my $other_val = ($pair_val == 1) ? 2 : 1; + + my $other_read_name = join("/", $core, $other_val); + + my $got_aligned_read = $aligned_reads_href->{$read_name} || 0; + my $got_other_aligned_read = $aligned_reads_href->{$other_read_name} || 0; + + #print "aligned_read: $read_name = $got_aligned_read\n"; + #print "other_read: $other_read_name = $got_other_aligned_read\n"; + + + unless ($aligned_reads_href->{$read_name} && $aligned_reads_href->{$other_read_name}) { + $print_record_flag = 1; + } + } + else { + # single read + + my $got_single = $aligned_reads_href->{$read_name} || 0; + #print "single: $read_name = $got_single\n"; + + unless ($aligned_reads_href->{$read_name}) { + $print_record_flag = 1; + } + } + } + + if ($print_record_flag) { + my $fastq_text = $record->get_fastq_record(); + print $fastq_text; + + $num_reads_output++; + } + else { + $num_reads_skipped++; + } + + #if ($num_reads_output > 5) { last; } + + } + + print STDERR "\n$fq_file: Total number of reads output: $num_reads_output, num skipped: $num_reads_skipped = " + . sprintf("%.2f", $num_reads_skipped/($num_reads_output+$num_reads_output)*100) . "\n"; + + return; +} + + + + diff --git a/99.scripts/trinity_utils/util/misc/cat_require_newlines.pl b/99.scripts/trinity_utils/util/misc/cat_require_newlines.pl new file mode 100644 index 0000000..2410e6b --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/cat_require_newlines.pl @@ -0,0 +1,18 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my @files = @ARGV; + +foreach my $file (@files) { + open (my $fh, $file) or die $!; + while (<$fh>) { + chomp; + print "$_\n"; + } + close $fh; +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/cdhit_examine_isoforms.pl b/99.scripts/trinity_utils/util/misc/cdhit_examine_isoforms.pl new file mode 100644 index 0000000..286c923 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/cdhit_examine_isoforms.pl @@ -0,0 +1,68 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "\n\tusage: $0 cd-hit.clstr\n\n" . + "try running cd-hit first like so:\n" . + "\tcd-hit-est -o cdhit -c 0.98 -i Trinity.fasta -p 1 -d 0 -b 3 -T 10\n\n"; + +my $cdhit_file = $ARGV[0] or die $usage; + +main: { + + my $num_bad_clusters = 0; + + my $cluster; + my @trans; + + open(my $fh, $cdhit_file) or die $!; + while (<$fh>) { + chomp; + if (/^>/) { + if (@trans) { + $num_bad_clusters += &examine_cluster($cluster, \@trans); + } + $cluster = $_; + @trans = (); + } + else { + push (@trans, $_); + } + } + close $fh; + + if (@trans) { + $num_bad_clusters += &examine_cluster($cluster, \@trans); + } + + print "Num bad clusters: $num_bad_clusters\n"; + + exit($num_bad_clusters); +} + +#### +sub examine_cluster { + my ($cluster, $trans_aref) = @_; + + my @trans = @$trans_aref; + + my %cluster_ids; + foreach my $tran (@trans) { + $tran =~ /TRINITY_(DN\d+)_/; + $cluster_ids{$1}++; + } + + my $num_clusters = scalar (keys %cluster_ids); + if ($num_clusters != 1) { + print STDERR "ERROR, got multiple clusters represented:\n" + . "$cluster\n" . join("\n", @trans) . "\n\n"; + return(1); + } + else { + return(0); + } +} + + + diff --git a/99.scripts/trinity_utils/util/misc/cdna_fasta_file_to_transcript_gtf.pl b/99.scripts/trinity_utils/util/misc/cdna_fasta_file_to_transcript_gtf.pl new file mode 100644 index 0000000..fbdeb54 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/cdna_fasta_file_to_transcript_gtf.pl @@ -0,0 +1,44 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use Fasta_reader; + +my $usage = "usage: $0 fasta_file\n\n"; + +my $fasta_file = $ARGV[0] or die $usage; + +my $fasta_reader = new Fasta_reader($fasta_file); + + +while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + + my $contig_acc = $acc; + + $acc =~ s/;/_/; + + my $seq = $seq_obj->get_sequence(); + + my $seq_len = length($seq); + + + print join("\t", $contig_acc, ".", "transcript", 1, $seq_len, ".", "+", ".", + "gene_id \"g.$acc\"; transcript_id \"t.$acc\";") . "\n"; + + print join("\t", $contig_acc, ".", "exon", 1, $seq_len, ".", "+", ".", + "gene_id \"g.$acc\"; transcript_id \"t.$acc\";") . "\n"; + + print "\n"; +} + + +exit(0); + + diff --git a/99.scripts/trinity_utils/util/misc/check_chrysalis_graph_reciprocal_edges.pl b/99.scripts/trinity_utils/util/misc/check_chrysalis_graph_reciprocal_edges.pl new file mode 100644 index 0000000..e3dfc66 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/check_chrysalis_graph_reciprocal_edges.pl @@ -0,0 +1,53 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "\n\n\tusage: $0 graphFromIwormFasta.out\n\n"; + +my $file = $ARGV[0] or die $usage; + +main: { + + my %graph; + + open(my $fh, $file) or die $!; + while (<$fh>) { + chomp; + my @x = split(/\t/); + + my $node_id = shift @x; + my $spacer = shift @x; + + foreach my $other_node (@x) { + $graph{$node_id}->{$other_node}++; + } + } + close $fh; + + + my $found_missing_recip = 0; + foreach my $node (keys %graph) { + + foreach my $other_node (keys %{$graph{$node}}) { + + if (! exists $graph{$other_node}->{$node}) { + $found_missing_recip = 1; + + print STDERR "Error, have $node\->$other_node, but missing $other_node\->$node\n"; + + } + } + } + + if ($found_missing_recip) { + die "Error, missing recips found.\n"; + } + else { + print STDERR "\n\nAll good. :-)\n\n"; + + exit(0); + } + + +} diff --git a/99.scripts/trinity_utils/util/misc/check_fastQ_pair_ordering.pl b/99.scripts/trinity_utils/util/misc/check_fastQ_pair_ordering.pl new file mode 100644 index 0000000..b53facf --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/check_fastQ_pair_ordering.pl @@ -0,0 +1,55 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fastq_reader; + +my $usage = "usage: $0 left.fq right.fq\n\n"; + +my $left_fq = $ARGV[0] or die $usage; +my $right_fq = $ARGV[1] or die $usage; + +main: { + + my $left_fq_reader = new Fastq_reader($left_fq); + my $right_fq_reader = new Fastq_reader($right_fq); + + my $ok_counter = 0; + my $error_counter = 0; + + while (my $left_fq_record = $left_fq_reader->next()) { + + my $right_fq_record = $right_fq_reader->next() or die "Error, no next record in $right_fq"; + + my $left_core_acc = $left_fq_record->get_core_read_name(); + my $left_full_acc = $left_fq_record->get_full_read_name(); + + my $right_core_acc = $right_fq_record->get_core_read_name(); + my $right_full_acc = $right_fq_record->get_full_read_name(); + + if ($left_core_acc eq $right_core_acc) { + $ok_counter++; + } + else { + $error_counter++; + print join("\t", $left_full_acc, $right_full_acc, $left_core_acc, $right_core_acc, "ERROR") . "\n\n"; + print STDERR "\r[$ok_counter ok, $error_counter error] "; + + + die "Error, found out-of-order pairing. Stopping now."; + + } + if ($ok_counter % 1000 == 0) { + print STDERR "\r[$ok_counter ok, $error_counter error] "; + } + + } + + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/chrys_graph_to_dot.pl b/99.scripts/trinity_utils/util/misc/chrys_graph_to_dot.pl new file mode 100644 index 0000000..6a1f1cb --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/chrys_graph_to_dot.pl @@ -0,0 +1,29 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 comp.raw.graph\n\n"; + +my $chrys_graph = $ARGV[0] or die $usage; + +main: { + + open (my $fh, $chrys_graph) or die "Error, cannot open file $chrys_graph"; + + print "digraph G {\n"; + + while (<$fh>) { + chomp; + my ($id, $prev_id, @rest) = split(/\t/); + if (defined ($id) && defined($prev_id)) { + print " $prev_id->$id\n"; + } + } + print "}\n"; + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/collate_fqs.pl b/99.scripts/trinity_utils/util/misc/collate_fqs.pl new file mode 100644 index 0000000..de224c5 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/collate_fqs.pl @@ -0,0 +1,143 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use Getopt::Long qw(:config posix_default no_ignore_case bundling pass_through); + +use FindBin; +use lib("$FindBin::Bin/../../PerlLib"); +use Fastq_reader; +use Data::Dumper; + +my $help_flag; + +my $usage = <<__EOUSAGE__; + +############################################################################################# +# +# --samples_file samples.txt +# +# --output_prefix outputs will be named .left.fq and .right.fq +# +############################################################################################# + + + +__EOUSAGE__ + + ; + + +my $samples_file; +my $output_prefix; + +&GetOptions ( 'h' => \$help_flag, + 'samples_file=s' => \$samples_file, + 'output_prefix=s' => \$output_prefix, + ); + +unless ($samples_file && $output_prefix) { + die $usage; +} + +main: { + + my @samples = &parse_samples_file($samples_file); + + # init fq readers + foreach my $sample (@samples) { + my $left_fq = $sample->{left_fq}; + my $left_fq_reader = new Fastq_reader($left_fq); + $sample->{left_fq_reader} = $left_fq_reader; + + my $right_fq = $sample->{right_fq}; + my $right_fq_reader = new Fastq_reader($right_fq); + $sample->{right_fq_reader} = $right_fq_reader; + } + + # open output files: + my $left_fq_outfile = "$output_prefix.left.fq"; + open(my $left_ofh, ">$left_fq_outfile") or die "Error, cannot write to $left_fq_outfile"; + + my $right_fq_outfile = "$output_prefix.right.fq"; + open(my $right_ofh, ">$right_fq_outfile") or die "Error, cannot write to $right_fq_outfile"; + + my $samples_remaining = scalar(@samples); + + my $counter = 0; + while($samples_remaining != 0) { + + $samples_remaining = 0; + + foreach my $sample (@samples) { + + if ($sample->{done}) { next; } + + my $sample_name = $sample->{sample_name}; + + my $left_fq_reader = $sample->{left_fq_reader}; + my $left_fq_entry = $left_fq_reader->next(); + + my $right_fq_reader = $sample->{right_fq_reader}; + my $right_fq_entry = $right_fq_reader->next(); + + if ($left_fq_entry xor $right_fq_entry) { + confess "Error, found left_fq entry but not right_fq entry: " . Dumper([$sample, $left_fq_entry, $right_fq_entry]); + } + elsif ($left_fq_entry && $right_fq_entry) { + + print $left_ofh $left_fq_entry->get_fastq_record(); + print $right_ofh $right_fq_entry->get_fastq_record(); + + $samples_remaining = 1; + + $counter++; + + if ($counter % 100000 == 0) { + print STDERR "\r[$counter] "; + } + + } + else { + $sample->{done} = 1; + $sample->{left_fq_reader}->finish(); + $sample->{right_fq_reader}->finish(); + } + + } + + } # end of while samples remaining + + close $left_ofh; + close $right_ofh; + + print STDERR "\n\nDone. Output $counter fastq records per file.\n"; + + exit(0); + } + + + + +#### +sub parse_samples_file { + my ($samples_file) = @_; + + my @samples; + + open(my $fh, $samples_file) or die "Error, cannot open file: $samples_file"; + while(<$fh>) { + chomp; + my ($sample_name, $left_fq, $right_fq) = split(/\t/); + push (@samples, { sample_name => $sample_name, + left_fq => $left_fq, + right_fq => $right_fq, + } ); + } + + close $fh; + + return(@samples); +} + diff --git a/99.scripts/trinity_utils/util/misc/combined_nameSorted_to_dup_pairs_removed.pl b/99.scripts/trinity_utils/util/misc/combined_nameSorted_to_dup_pairs_removed.pl new file mode 100644 index 0000000..d0e6129 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/combined_nameSorted_to_dup_pairs_removed.pl @@ -0,0 +1,122 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 combined_nameSorted.sam\n\n"; + +my $combined_nameSorted_sam = $ARGV[0] or die $usage; + + +main: { + + my $conglom_file = "$combined_nameSorted_sam.__conglom_tmp"; + + unless (-s $conglom_file) { + + open (my $ofh, ">$conglom_file") or die "Error, cannot write to $conglom_file"; + + my @paired; + + my $prev_acc = ""; + + open (my $fh, $combined_nameSorted_sam) or die $!; + while (<$fh>) { + chomp; + my @x = split(/\t/); + if ($x[0] ne $prev_acc) { + if (scalar(@paired) == 2) { + &process_pair($ofh, \@paired); + } + @paired = (); + } + $prev_acc = $x[0]; + push (@paired, [@x]); + } + + ## get last ones + if (scalar(@paired) == 2) { + &process_pair($ofh, \@paired); + } + + close $ofh; + } + + + + my $cmd = "sort -k1,1 -k2,2n -k3,3 -k4,4n $conglom_file > $conglom_file.sorted"; + &process_cmd($cmd) unless (-s "$conglom_file.sorted"); + + my $dups_removed_file = "$conglom_file.dups_removed.sam"; + unless (-s $dups_removed_file) { + open (my $ofh, ">$dups_removed_file") or die $!; + open (my $fh, "$conglom_file.sorted") or die $!; + print STDERR "-writing $dups_removed_file\n"; + my $prev_contig_info = ""; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my $contig_info = join("\t", @x[0..3]); + + my $read_A_info = $x[4]; + my $read_B_info = $x[5]; + if ($contig_info ne $prev_contig_info) { + $read_A_info =~ s/$;/\t/g; + $read_B_info =~ s/$;/\t/g; + + print $ofh "$read_A_info\n$read_B_info\n"; + } + $prev_contig_info = $contig_info; + } + close $fh; + close $ofh; + } + + + + exit(0); + + +} + + +#### +sub process_pair { + my ($ofh, $pairs_aref) = @_; + + my ($read_A, $read_B) = @$pairs_aref; + + if ($read_A->[2] gt $read_B->[2]) { + ## swap 'em + ($read_A, $read_B) = ($read_B, $read_A); + } + + my $contig_A = $read_A->[2]; + my $pos_A = $read_A->[3]; + + my $contig_B = $read_B->[2]; + my $pos_B = $read_B->[3]; + + my $read_A_text = join("$;", @$read_A); + my $read_B_text = join("$;", @$read_B); + + print $ofh join("\t", $contig_A, $pos_A, $contig_B, $pos_B, $read_A_text, $read_B_text) . "\n"; + + return; + +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} diff --git a/99.scripts/trinity_utils/util/misc/compare_FL_stats.pl b/99.scripts/trinity_utils/util/misc/compare_FL_stats.pl new file mode 100644 index 0000000..9c622b7 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/compare_FL_stats.pl @@ -0,0 +1,70 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "\n\tusage: $0 A.FL_selected B.FL_selected [RESTRICT_TO_FAILURES]\n\n"; + +my $A_stats_file = $ARGV[0] or die $usage; +my $B_stats_file = $ARGV[1] or die $usage; + +my $RESTRICT_TO_FAILURES = $ARGV[2] || 0; + +my $MIN_PER_LEN = 99; +my $MAX_PER_GAP = 1; + +main: { + + my %A_stats = &parse_stats($A_stats_file); + my %B_stats = &parse_stats($B_stats_file); + + my %trans = map { + $_ => 1 } (keys %A_stats, keys %B_stats); + + foreach my $trans_acc (keys %trans) { + + my $A_FL = $A_stats{$trans_acc} || "."; + + my $B_FL = $B_stats{$trans_acc} || "."; + + if ($RESTRICT_TO_FAILURES) { + unless ($A_FL eq "." || $B_FL eq ".") { + next; + } + } + + my ($trans, $gene) = split(/;/, $trans_acc); + + print join("\t", $gene, $trans, $trans_acc, $A_FL, $B_FL) . "\n"; + + } + + exit(0); + +} + +#### +sub parse_stats { + my ($stats_file) = @_; + + my %trans_to_stats; + + open (my $fh, $stats_file) or die $!; + while (<$fh>) { + chomp; + + my $line = $_; + + my @x = split(/\t/); + + + my $trans = $x[0]; + my $reco = $x[1]; + + $trans_to_stats{$trans} = $reco; + + } + close $fh; + + return (%trans_to_stats); +} + diff --git a/99.scripts/trinity_utils/util/misc/compare_bflies.pl b/99.scripts/trinity_utils/util/misc/compare_bflies.pl new file mode 100644 index 0000000..896cac2 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/compare_bflies.pl @@ -0,0 +1,95 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Cwd; +use File::Basename; + +my $usage = "usage: $0 cNumb.graph\n\n"; + +my $comp = $ARGV[0] or die $usage; + +my $TRINITY_HOME = $ENV{TRINITY_HOME} or die "Error, need env var TRINITY_HOME"; + +main: { + + + my $base_comp_name = basename($comp); + + my $cwd = cwd(); + + unless ($comp =~ /^\//) { + $comp = "$cwd/$comp"; + + } + + + + + { ## run old butterfly + + my $workdir = "$cwd/__$base_comp_name.oldBfly_dir"; + + mkdir($workdir) or die "Error, cannot mkdir $workdir"; + chdir $workdir or die $!; + + my $cmd = "ln -s $comp.reads .; ln -s $comp.out ."; + &process_cmd($cmd); + + $cmd = "java -Xmx4G -jar $TRINITY_HOME/Butterfly/prev_vers/Butterfly_r2013_08_14.jar -N 100000 -L 200 -F 500 -C " . basename($comp) . " --path_reinforcement_distance=75 --max_number_of_paths_per_node=10 -V 15 --stderr 2>&1 | tee log.txt"; + + open (my $ofh, ">bfly.cmd") or die $!; + print $ofh $cmd; + close $ofh; + + &process_cmd($cmd); + + } + + chdir $cwd or die $!; + + + { ## run new butterfly + + my $workdir = "$cwd/__$base_comp_name.newBfly_dir"; + + mkdir($workdir) or die $!; + chdir ($workdir) or die $!; + + my $cmd = "ln -s $comp.reads .; ln -s $comp.out ."; + &process_cmd($cmd); + + + $cmd = "java -Xmx4G -jar $TRINITY_HOME/Butterfly/Butterfly.jar -N 100000 -L 200 -F 500 -C " . basename($comp) . " --path_reinforcement_distance=75 -V 15 --stderr 2>&1 | tee log.txt"; + + open (my $ofh, ">bfly.cmd") or die $!; + print $ofh $cmd; + close $ofh; + + + &process_cmd($cmd); + + + } + + + exit(0); +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + + if ($ret) { + die "Error, CMD: $cmd died with ret $ret"; + } + + return; +} + + diff --git a/99.scripts/trinity_utils/util/misc/component_to_graph_dot.pl b/99.scripts/trinity_utils/util/misc/component_to_graph_dot.pl new file mode 100644 index 0000000..225044c --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/component_to_graph_dot.pl @@ -0,0 +1,153 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "\n\nusage: $0 chrysalis_dir/ component_no (WELDS|SCAFF|BOTH)\n\n"; + +my $chrysalis_dir = $ARGV[0] or die $usage; +my $component_no = $ARGV[1]; +my $draw_type = $ARGV[2] or die $usage; + +unless (defined $component_no) { + die $usage; +} + +unless ($draw_type =~ /^(WELDS|SCAFF|BOTH)$/) { + die $usage; +} + +main: { + + my $graph_from_iworm_file = "$chrysalis_dir/GraphFromIwormFasta.out"; + + my %iworm_accs = &get_iworm_list_from_component($graph_from_iworm_file, $component_no); + + ## get the weld links: + my %welds = &parse_welds($graph_from_iworm_file, \%iworm_accs); + + ## get the scaffolding links: + my $scaffolding_file = "$chrysalis_dir/../iworm_scaffolds.txt"; + my %scaffolds = &parse_scaffolded_pairs($scaffolding_file, \%iworm_accs); + + ## write dot file + print "digraph {\n"; + + + my %iworm_to_id; + ## write node labels + my $counter = 0; + foreach my $iworm_acc (keys %iworm_accs) { + $counter++; + my $iworm_id = $iworm_to_id{$iworm_acc} = $counter; + print "\t$iworm_id [label=\"$iworm_acc\"];\n"; + } + + if ($draw_type =~ /WELDS|BOTH/) { + ## add weld links + foreach my $iworm_acc (keys %welds) { + my $iworm_id = $iworm_to_id{$iworm_acc}; + foreach my $welded_iworm_acc (keys %{$welds{$iworm_acc}}) { + my $welded_id = $iworm_to_id{$welded_iworm_acc}; + print "\t$iworm_id" . "->" . "$welded_id [color=\"blue\"];\n"; + } + } + } + + if ($draw_type =~ /SCAFF|BOTH/) { + ## add scaffolding links + foreach my $iworm_acc (keys %scaffolds) { + my $iworm_id = $iworm_to_id{$iworm_acc}; + unless (defined $iworm_id) { next; } + + foreach my $scaffolded_iworm_acc (keys %{$scaffolds{$iworm_acc}}) { + my $scaffolded_id = $iworm_to_id{$scaffolded_iworm_acc}; + unless (defined $scaffolded_id) { next; } # not all scaffolded entries were chosen for clustering if read support was insufficient + + print "\t$iworm_id" . "->" . "$scaffolded_id [color=\"green\"];\n"; + } + } + } + + print "}\n"; + + exit(0); + + +} + + +#### +sub parse_welds { + my ($graph_file, $iworm_accs_href) = @_; + + my %welds; + + open (my $fh, $graph_file) or die "Error, cannot oepn file $graph_file"; + while (<$fh>) { + chomp; + if (/^\#Welding: >(a\d+;\d+)_\S+ to >(a\d+;\d+)_/) { + my $iworm_acc_A = $1; + my $iworm_acc_B = $2; + + ($iworm_acc_A, $iworm_acc_B) = sort ($iworm_acc_A, $iworm_acc_B); + + if ($iworm_accs_href->{$iworm_acc_A} || $iworm_accs_href->{$iworm_acc_B}) { # really should only have situations where they're both included + + $welds{$iworm_acc_A}->{$iworm_acc_B} = 1; + } + } + } + + close $fh; + + return(%welds); +} + + +#### +sub get_iworm_list_from_component { + my ($graph_file, $component_no) = @_; + + my %iworm_accs; + + open (my $fh, $graph_file) or die "Error, cannot open file $graph_file"; + while (<$fh>) { + if (/^>Component_$component_no /) { + /\[iworm>(a\d+;\d+)_/ or die "Error, cannot parse iworm acc from $_ "; + my $iworm_acc = $1; + + $iworm_accs{$iworm_acc} = 1; + } + } + close $fh; + + + return(%iworm_accs); +} + + +#### +sub parse_scaffolded_pairs { + my ($scaffolding_file, $iworm_accs_href) = @_; + + my %scaffolded_pairs; + + open (my $fh, $scaffolding_file) or die "Error, cannot open file $scaffolding_file"; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my ($iworm_acc_A, $iworm_acc_B) = ($x[0], $x[2]); + + ($iworm_acc_A, $iworm_acc_B) = sort ($iworm_acc_A, $iworm_acc_B); + + if ($iworm_accs_href->{$iworm_acc_A} || $iworm_accs_href->{$iworm_acc_B}) { + + $scaffolded_pairs{$iworm_acc_A}->{$iworm_acc_B} = 1; + } + } + close $fh; + + return(%scaffolded_pairs); +} + diff --git a/99.scripts/trinity_utils/util/misc/contig_ExN50_statistic.pl b/99.scripts/trinity_utils/util/misc/contig_ExN50_statistic.pl new file mode 100644 index 0000000..a3bf23d --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/contig_ExN50_statistic.pl @@ -0,0 +1,200 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; +use File::Basename; + + +my $usage = "usage: $0 isoform.EXPR.matrix Trinity.fasta [by=transcript|gene (default:transcript)]\n\n" + . "\t note, use the isoform.EXPR.matrix file regardiess of wehther you choose transcript | gene feature type to explore.\n\n"; + + +my $matrix_file = $ARGV[0] or die $usage; +my $fasta_file = $ARGV[1] or die $usage; +my $by_feature_type = $ARGV[2] || "transcript"; + + +unless (-s $matrix_file) { + die "Error, cannot locate matrix file: $matrix_file"; +} +unless (-s $fasta_file) { + die "Error, cannot locate fasta file: $fasta_file"; +} + +unless ($by_feature_type =~ /^(transcript|gene)$/) { + die "Error, cannot discern feature type: [$by_feature_type] "; +} + +my %trans_lengths; +{ + my $fasta_reader = new Fasta_reader($fasta_file); + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + my $sequence = $seq_obj->get_sequence(); + + my $seq_len = length($sequence); + + $trans_lengths{$acc} = $seq_len; + } +} + +open (my $fh, $matrix_file) or die $!; +my $header = <$fh>; + +my %gene_to_trans; + +my $sum_expr = 0; + + +my $feature_type = "transcript"; + +while (<$fh>) { + chomp; + my @x = split(/\t/); + my $acc = shift @x; # gene accession + my $max_expr = 0; + my $trans_sum_expr = 0; + while (@x) { + my $expr = shift @x; + + $trans_sum_expr += $expr; + $sum_expr += $expr; + + if ($expr > $max_expr) { + $max_expr = $expr; + } + } + + my $seq_len = $trans_lengths{$acc} or die "Error, no seq length for acc: $acc. Be sure to give the isoform.TPM expression matrix as input parameter"; + + my $gene_id = $acc; + + + if ($by_feature_type =~ /gene/) { + if ($acc =~ /^(\S+)_i\d+/) { + $gene_id = $1; + $feature_type = "gene"; + } + else { + die "Error, by_feature_type is gene, but cannot extract gene_id from $acc "; + } + } + + push (@{$gene_to_trans{$gene_id}}, { acc => $acc, + len => $seq_len, + sum_expr => $trans_sum_expr, + max_expr => $max_expr, + }); + +} + +my @genes; + +## make expression weighted gene length +foreach my $gene (keys %gene_to_trans) { + my @trans_structs = @{$gene_to_trans{$gene}}; + + my $sum_expr = 0; + my $sum_expr_n_len = 0; + my $max_expr = 0; + foreach my $trans_struct (@trans_structs) { + my $len = $trans_struct->{len}; + my $expr = $trans_struct->{sum_expr} || 1; + $sum_expr_n_len += $len * $expr; + $sum_expr += $expr; + + my $m_expr = $trans_struct->{max_expr}; + if ($m_expr > $max_expr) { + $max_expr = $m_expr; + } + + } + + my $gene_len = $sum_expr_n_len / $sum_expr; + + push (@genes, { acc => $gene, + sum_expr => $sum_expr, + len => $gene_len, + max_expr => $max_expr } ); +} + + +@genes = reverse sort { $a->{sum_expr} <=> $b->{sum_expr} + || + $a->{len} <=> $b->{len} } @genes; + + + +## write output table + + +my $E_file = basename($matrix_file) . ".by-$feature_type.E-inputs"; +open (my $ofh, ">$E_file") or die $!; +print $ofh join("\t", "#Ex", "acc", "length", "max_expr_over_samples", "sum_expr_over_samples") . "\n"; + +print "Ex\tExN50\tnum_${feature_type}s\n"; + +my $prev_pct = 0; +my $sum = 0; +my @captured; +while (@genes) { + + my $t = shift @genes; + + $sum += $t->{sum_expr}; + + my $pct = int($sum/$sum_expr * 100); + + print $ofh join("\t", $pct, $t->{acc}, int($t->{len}), sprintf("%.3f", $t->{max_expr}), sprintf("%.3f", $t->{sum_expr})) . "\n"; + + if ($prev_pct > 0 && $pct > $prev_pct) { + + + my $N50 = int(&calc_N50(@captured)); + my $num_trans = scalar(@captured); + + print "$prev_pct\t$N50\t$num_trans\n"; + } + + $prev_pct = $pct; + + push (@captured, $t); +} + +# do last one + +my $N50 = &calc_N50(@captured); +my $num_genes = scalar(@captured); +print "100\t$N50\t$num_genes\n"; + + +exit(0); + + +#### +sub calc_N50 { + my @entries = @_; + + @entries = reverse sort {$a->{len}<=>$b->{len}} @entries; + + my $sum_len = 0; + foreach my $entry (@entries) { + $sum_len += $entry->{len}; + } + + my $loc_sum = 0; + foreach my $entry (@entries) { + $loc_sum += $entry->{len}; + if ($loc_sum / $sum_len * 100 >= 50) { + return($entry->{len}); + } + } + + return(-1); # error +} + diff --git a/99.scripts/trinity_utils/util/misc/convert_fasta_identifiers_for_FL_analysis.pl b/99.scripts/trinity_utils/util/misc/convert_fasta_identifiers_for_FL_analysis.pl new file mode 100644 index 0000000..14d8fee --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/convert_fasta_identifiers_for_FL_analysis.pl @@ -0,0 +1,44 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use lib ($ENV{EUK_MODULES}); +use Fasta_reader; + + +my $usage = "usage: $0 file.fasta\n\n"; + +my $fasta_file = $ARGV[0] or die $usage; + +main: { + + my $fasta_reader = new Fasta_reader($fasta_file); + + my %seen; + + while (my $seq_obj = $fasta_reader->next()) { + + my $header = $seq_obj->get_header(); + + my $seq = $seq_obj->get_sequence(); + + my ($trans_id, $gene_id, @rest) = split(/\s+/, $header); + + + if ($seen{$seq}) { + next; + } + + $seen{$seq} = 1; + + + print ">$trans_id;$gene_id\n$seq\n"; + } + + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/count_N50_given_MIN_FPKM_threshold.pl b/99.scripts/trinity_utils/util/misc/count_N50_given_MIN_FPKM_threshold.pl new file mode 100644 index 0000000..023226e --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/count_N50_given_MIN_FPKM_threshold.pl @@ -0,0 +1,84 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + + +my $usage = "\n\nusage: $0 RSEM.genes.results\n\n"; + +my $rsem_file = $ARGV[0] or die $usage; + +main: { + + my @entries; + + open (my $fh, $rsem_file) or die $!; + my $header = <$fh>; + my $max_fpkm = 0; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my $len = $x[2]; + my $fpkm = $x[6]; + + push (@entries, { len => $len, + fpkm => $fpkm, + }); + + + if ($fpkm > $max_fpkm) { + $max_fpkm = $fpkm; + } + } + close $fh; + + print join("\t", "min_fpkm", "cum_seq_len", "partial_sum_len", "N50_len", "num_entries") . "\n"; + ; + &N50(0, \@entries); + + my $min_fpkm = 0.01; + + while ($min_fpkm < $max_fpkm) { + + @entries = grep { $_->{fpkm} >= $min_fpkm } @entries; + + &N50($min_fpkm, \@entries); + + $min_fpkm *= 2; + } + +} + +sub N50 { + my ($min_fpkm, $entries_aref) = @_; + + my @entries = @$entries_aref; + @entries = reverse sort {$a->{len}<=>$b->{len}} @entries; + + my $num_entries = scalar(@entries); + + my $cum_seq_len = 0; + foreach my $entry (@entries) { + $cum_seq_len += $entry->{len}; + } + + my $half_cum_len = $cum_seq_len / 2; + + my $n50_len = "NA"; + my $partial_sum_len = 0; + foreach my $entry (@entries) { + my $len = $entry->{len}; + $partial_sum_len += $len; + + if ($partial_sum_len >= $half_cum_len) { + $n50_len = $len; + last; + } + } + + + print join("\t", $min_fpkm, int($cum_seq_len+0.5), int($partial_sum_len+0.5), $n50_len, $num_entries) . "\n"; + + return; +} + diff --git a/99.scripts/trinity_utils/util/misc/count_features_given_MIN_FPKM_threshold.pl b/99.scripts/trinity_utils/util/misc/count_features_given_MIN_FPKM_threshold.pl new file mode 100644 index 0000000..6a7b968 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/count_features_given_MIN_FPKM_threshold.pl @@ -0,0 +1,44 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 expr.RSEM\n\n"; + +my $expr_file = $ARGV[0] or die $usage; + +open (my $fh, $expr_file) or die $!; +my $header = <$fh>; + +my @fpkms; +while (<$fh>) { + chomp; + my @x = split(/\t/); + my $fpkm = $x[6]; + push (@fpkms, $fpkm); +} + +@fpkms = reverse sort {$a<=>$b} @fpkms; + +my $min_fpkm_thresh = int($fpkms[0] + 0.5); +my $num_features = 1; + +print "neg_min_fpkm\tnum_features\n"; + +shift @fpkms; +while (@fpkms) { + + my $fpkm = shift @fpkms; + $fpkm = int($fpkm+0.5); + + if ($fpkm < $min_fpkm_thresh) { + print "" . (-1*$min_fpkm_thresh) . "\t$num_features\n"; + $min_fpkm_thresh = $fpkm; + + } + $num_features++; +} + +print "" . (-1*$min_fpkm_thresh) . "\t$num_features\n"; + +exit(0); diff --git a/99.scripts/trinity_utils/util/misc/count_iso_per_gene_dist.pl b/99.scripts/trinity_utils/util/misc/count_iso_per_gene_dist.pl new file mode 100644 index 0000000..adc50b8 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/count_iso_per_gene_dist.pl @@ -0,0 +1,38 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + + +my %counter; + +while (<>) { + if (/^>/) { + if (/^>(.*c\d+_g\d+)/) { + $counter{$1}++; + } + elsif (/^>(.*comp\d+_c\d+)/) { + # older format + $counter{$1}++; + } + else { + die "Error, dont recognize formatting of accession in line: $_"; + } + } +} + +my %count_counter; +for my $val (values %counter) { + $count_counter{$val}++; +} + +print join("\t", "#iso_per_gene", "num_genes") . "\n"; + +for my $count (sort {$a<=>$b} keys %count_counter) { + my $val = $count_counter{$count}; + + print join("\t", $count, $val) . "\n"; +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/count_matrix_features_given_MIN_TPM_threshold.pl b/99.scripts/trinity_utils/util/misc/count_matrix_features_given_MIN_TPM_threshold.pl new file mode 100644 index 0000000..e6c1680 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/count_matrix_features_given_MIN_TPM_threshold.pl @@ -0,0 +1,51 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 tpm.matrix\n\n"; + + +my $matrix_file = $ARGV[0] or die $usage; + +open (my $fh, $matrix_file) or die $!; +my $header = <$fh>; + +my @tpms; +while (<$fh>) { + chomp; + my @x = split(/\t/); + shift @x; # gene accession + my $max_tpm = shift @x; + while (@x) { + my $tpm = shift @x; + if ($tpm > $max_tpm) { + $max_tpm = $tpm; + } + } + push (@tpms, $max_tpm); +} + +@tpms = reverse sort {$a<=>$b} @tpms; + +my $min_tpm_thresh = int($tpms[0]); +my $num_features = 1; + +print "neg_min_tpm\tnum_features\n"; + +shift @tpms; +while (@tpms) { + + my $tpm = shift @tpms; + + if ($tpm < $min_tpm_thresh) { + print "" . (-1*$min_tpm_thresh) . "\t$num_features\n"; + $min_tpm_thresh = int($tpm); + + } + $num_features++; +} + +print "$min_tpm_thresh\t$num_features\n"; + +exit(0); diff --git a/99.scripts/trinity_utils/util/misc/count_number_fasta_seqs.pl b/99.scripts/trinity_utils/util/misc/count_number_fasta_seqs.pl new file mode 100644 index 0000000..54655cd --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/count_number_fasta_seqs.pl @@ -0,0 +1,25 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +while (<>) { + my $filename = $_; + chomp $filename; + + my $count = 0; + + open (my $fh, $filename) or die "Error, cannot open file $filename"; + while (<$fh>) { + if (/^>/) { + $count++; + } + } + print "$count\t$filename\n"; + close $fh; + +} + + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/count_trans_per_component.pl b/99.scripts/trinity_utils/util/misc/count_trans_per_component.pl new file mode 100644 index 0000000..d3bd65c --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/count_trans_per_component.pl @@ -0,0 +1,54 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; + +my $usage = "usage: $0 bfly.A.fasta [bfly.B.fasta ...]\n\n"; + +my @files = @ARGV or die $usage; + +main: { + + my %data; + + foreach my $file (@files) { + open (my $fh, $file) or die "Error, cannot open file $file"; + while (<$fh>) { + + unless (/^>/) { next; } + + my $comp_id; + if (/>\S*(c\d+\.graph_c\d+)_seq\d+/) { + $comp_id = $1; + } + elsif (/^>\S*(comp\d+_c\d+)_seq\d+/) { + $comp_id = $1; + } + else { + die "Error, couldn't extract component identifier from $_"; + } + + $data{$comp_id}->{$file}++; + } + } + + ## output data + print join("\t", "#component", @files) . "\n"; + foreach my $component (keys %data) { + + print "$component"; + foreach my $file (@files) { + my $count = $data{$component}->{$file} || 0; + print "\t$count"; + } + print "\n"; + } + + exit(0); +} + + + + diff --git a/99.scripts/trinity_utils/util/misc/decode_SAM_flag_value.pl b/99.scripts/trinity_utils/util/misc/decode_SAM_flag_value.pl new file mode 100644 index 0000000..bc9d305 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/decode_SAM_flag_value.pl @@ -0,0 +1,61 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 flag_number\n\n"; + +my $flag = $ARGV[0] or die $usage; + + +my @tokens; + +my $pair_flag = ($flag & 0x0001) ? "PAIRED" : "unpaired"; +push (@tokens, $pair_flag); + +my $query_unmapped_flag = ($flag & 0x0004) ? "query_unmapped" : "QUERY_MAPPED"; +push (@tokens, $query_unmapped_flag); + +if ($query_unmapped_flag eq "QUERY_MAPPED") { + my $query_strand = ($flag & 0x0010) ? "QUERY_REVERSE_STRAND" : "QUERY_FORWARD_STRAND"; + push (@tokens, $query_strand); +} + + +if ($pair_flag eq "PAIRED") { + + my $mapped_proper_pair_flag = ($flag & 0x0002) ? "MAPPED_PROPER_PAIR" : "not_mapped_proper_pair"; + push (@tokens, $mapped_proper_pair_flag); + + my $mate_mapped = ($flag & 0x0008) ? "mate_unmapped" : "MATE_MAPPED"; + push (@tokens, $mate_mapped); + + if ($mate_mapped eq "MATE_MAPPED") { + my $mate_strand = ($flag & 0x0020) ? "MATE_REVERSE_STRAND" : "MATE_FORWARD_STRAND"; + push (@tokens, $mate_strand); + } + + my $first_in_pair = ($flag & 0x0040) ? "FIRST_IN_PAIRx40" : "SECOND_IN_PAIRx40"; + push (@tokens, $first_in_pair); + + my $second_in_pair = ($flag & 0x0080) ? "SECOND_IN_PAIRx80" : "FIRST_IN_PAIRx80"; + push (@tokens, $second_in_pair); +} + + +my $primary_flag = ($flag & 0x0100) ? "notprimary" : "PRIMARY"; +push (@tokens, $primary_flag); + +my $fails_quality_checks = ($flag & 0x0200) ? "FAILED" : ""; +push (@tokens, $fails_quality_checks) if $fails_quality_checks; + +my $pcr_op_duplicate = ($flag & 0x0400) ? "PCROPDUP" : ""; +push (@tokens, $pcr_op_duplicate) if $pcr_op_duplicate; + +print "flag($flag) = " . join ("...", @tokens) . "\n"; + + + +exit(0); + + diff --git a/99.scripts/trinity_utils/util/misc/describe_SAM_read_flag_info.pl b/99.scripts/trinity_utils/util/misc/describe_SAM_read_flag_info.pl new file mode 100644 index 0000000..8722ba9 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/describe_SAM_read_flag_info.pl @@ -0,0 +1,95 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 file.sam [include_SAM=0]\n\n"; + +my $SAM_FILE = $ARGV[0] or die $usage; +my $INCLUDE_SAM = $ARGV[1]; + + +my $fh; +if ($SAM_FILE eq "-") { + $fh = \*STDIN; +} +else { + if ($SAM_FILE =~ /\.bam$/) { + open ($fh, "samtools view $SAM_FILE |") or die "Error, $!"; + } + else { + open ($fh, $SAM_FILE) or die "Error, cannot open file $SAM_FILE"; + } +} + + +while (<$fh>) { + chomp; + + my $line = $_; + + if (/^\@/) { next; } # skip header + my @x = split(/\t/); + + my $read_acc = $x[0]; + my $flag = $x[1]; + my $qual_string = $x[10]; + my $target = $x[2]; + + my @tokens; + + my $pair_flag = ($flag & 0x0001) ? "PAIRED" : "notpaired"; + push (@tokens, $pair_flag); + + my $query_unmapped_flag = ($flag & 0x0004) ? "qun" : "QM"; + push (@tokens, $query_unmapped_flag); + + if ($query_unmapped_flag eq "QM") { + my $query_strand = ($flag & 0x0010) ? "QREV" : "QFWD"; + push (@tokens, $query_strand); + } + + + if ($pair_flag eq "PAIRED") { + + my $mapped_proper_pair_flag = ($flag & 0x0002) ? "MAPPEDPROPERPAIR" : "notmappedproperpair"; + push (@tokens, $mapped_proper_pair_flag); + + my $mate_mapped = ($flag & 0x0008) ? "mateunmapped" : "MATEMAPPED"; + push (@tokens, $mate_mapped); + + if ($mate_mapped eq "MM") { + my $mate_strand = ($flag & 0x0020) ? "MREV" : "MFWD"; + push (@tokens, $mate_strand); + } + + my $first_in_pair = ($flag & 0x0040) ? "FIRSTx40" : "SECONDx40"; + push (@tokens, $first_in_pair); + + my $second_in_pair = ($flag & 0x0080) ? "SECONDx80" : "FIRSTx80"; + push (@tokens, $second_in_pair); + } + + + my $primary_flag = ($flag & 0x0100) ? "notprimary" : "PRIMARY"; + push (@tokens, $primary_flag); + + my $fails_quality_checks = ($flag & 0x0200) ? "FAILED" : ""; + push (@tokens, $fails_quality_checks) if $fails_quality_checks; + + my $pcr_op_duplicate = ($flag & 0x0400) ? "PCROPDUP" : ""; + push (@tokens, $pcr_op_duplicate) if $pcr_op_duplicate; + + if ($INCLUDE_SAM) { + print $line; + } + else { + print "$read_acc\t$target"; + } + print "\t" . join ("_", @tokens) . "\n"; + +} + +exit(0); + + diff --git a/99.scripts/trinity_utils/util/misc/determine_RF_strand_specificity.pl b/99.scripts/trinity_utils/util/misc/determine_RF_strand_specificity.pl new file mode 100644 index 0000000..5ec4f6b --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/determine_RF_strand_specificity.pl @@ -0,0 +1,81 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "\n\n\tusage: $0 transcript_aligned.bam top_percent=10\n\n"; + +my $bam_file = $ARGV[0] or die $usage; +my $top_percent = $ARGV[1] || 10; + +main: { + + my %transcript_to_orients; + + my $sam_reader = new SAM_reader($bam_file); + print STDERR "-parsing file: $bam_file\n"; + while (my $sam_entry = $sam_reader->get_next()) { + + if ($sam_entry->is_proper_pair() && $sam_entry->is_first_in_pair()) { + + my $orient = $sam_entry->get_query_strand(); + my $trans_name = $sam_entry->get_scaffold_name(); + + #print STDERR "$trans_name\t$orient\n"; + + $transcript_to_orients{$trans_name}->{$orient}++; + } + } + print STDERR "-done parsing file, examining orientations of reads.\n"; + + ## sum them up. + my @transcripts = keys %transcript_to_orients; + foreach my $transcript (@transcripts) { + + my $orient_plus = $transcript_to_orients{$transcript}->{'+'} || 0; + my $orient_minus = $transcript_to_orients{$transcript}->{'-'} || 0; + + + my $total_reads = $orient_plus + $orient_minus; + + $transcript_to_orients{$transcript}->{'transcript'} = $transcript; + $transcript_to_orients{$transcript}->{'total_reads'} = $total_reads; + } + + #### + my @structs = values %transcript_to_orients; + @structs = reverse sort {$a->{total_reads}<=>$b->{total_reads}} @structs; + + my $num_transcripts = scalar(@structs); + my $max_index = int($top_percent/100 * $num_transcripts); + + # header + print join("\t", "#transcript", "minus_strand_1stReads", "plus_strand_1stReads", "total_reads", "pct_RF_SS") . "\n"; + + for (0..$max_index) { + my $struct = shift @structs; + + + my ($plus, $minus, $total_reads, $transcript) = ($struct->{'+'}, + $struct->{'-'}, + $struct->{'total_reads'}, + $struct->{'transcript'}); + unless ($plus) { $plus = 0; } + unless ($minus) { $minus = 0; } + + # assuming RF + my $pct_total = $minus / $total_reads * 100; + + print join("\t", $transcript, $minus, $plus, $total_reads, sprintf("%.1f", $pct_total)) . "\n"; + + } + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/ensure_paired_end_bam_file.pl b/99.scripts/trinity_utils/util/misc/ensure_paired_end_bam_file.pl new file mode 100644 index 0000000..b3cf9b9 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/ensure_paired_end_bam_file.pl @@ -0,0 +1,47 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use SAM_reader; +use SAM_entry; + +use Carp; +use Data::Dumper; + +my $usage = "usage: $0 file.sam number_records_to_check=10\n\n"; + + +my $sam_file = $ARGV[0] or die $usage; +my $num_records_to_check = $ARGV[1] || 10; + +main: { + + my $sam_reader = new SAM_reader($sam_file); + + my $num_records_checked = 0; + + while (my $sam_entry = $sam_reader->get_next()) { + + my $core_read_name = $sam_entry->get_read_name(); + + if (! $sam_entry->is_paired()) { + confess "ERROR, only paired reads should exist in bam file. Encountered unpaired read: " . Dumper($sam_entry); + } + + $num_records_checked++; + + if ($num_records_checked >= $num_records_to_check) { + last; + } + } + + print STDERR "paired read check for $sam_file is OK.\n"; + + exit(0); # all good! + +} + diff --git a/99.scripts/trinity_utils/util/misc/examine_iworm_FL_across_threads.pl b/99.scripts/trinity_utils/util/misc/examine_iworm_FL_across_threads.pl new file mode 100644 index 0000000..b5c8ff7 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/examine_iworm_FL_across_threads.pl @@ -0,0 +1,97 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my @files = ; + +my %pid_to_files; +foreach my $file (@files) { + $file =~ /pid_(\d+)/ or die "error, cannnot get pid from filename $file"; + my $pid = $1; + push (@{$pid_to_files{$pid}}, $file); +} + +{ + ## handle the mpi cases + @files = <*node_*contigs*pslx>; + for my $file (@files) { + push (@{$pid_to_files{"mpi"}}, $file); + } +} + + +foreach my $pid (keys %pid_to_files) { + my @files = @{$pid_to_files{$pid}}; + + print "$pid\t" . scalar(@files) . "\n"; + my $num_threads = scalar(@files); + + my %FL_acc_to_file; + my %FL_file_to_acc; + + foreach my $file (@files) { + + my $FL_file_counter = 0; + + my $fl_file = "$file.FL_selected"; + open (my $fh, $fl_file) or die $!; + while (<$fh>) { + my @x = split(/\t/); + my $acc = $x[13]; + + $FL_acc_to_file{$acc}->{$fl_file} = 1; + $FL_file_counter++; + + $FL_file_to_acc{$fl_file}->{$acc}++; + + } + close $fh; + + print "Num threads: $num_threads\t$fl_file\t$FL_file_counter\n"; + } + + + my $number_FL = scalar(keys %FL_acc_to_file); + print "Num threads: $num_threads\tTotal:\n$number_FL\n"; + + print "\n#" . join("\t", "file", "total_FL") . "\n"; + ## generate distribution of FL counts. + my %count_counter; + my %acc_counter; + foreach my $acc (keys %FL_acc_to_file) { + my @files = keys %{$FL_acc_to_file{$acc}}; + my $num_files = scalar @files; + $count_counter{$num_files}++; + $acc_counter{$acc} = $num_files; + } + + + print "\n#" . join("\t", "file", "non-redundant", "redundant") . "\n"; + foreach my $file (keys %FL_file_to_acc) { + my @accs = keys %{$FL_file_to_acc{$file}}; + my $num_total_accs = scalar(@accs); + + my $num_redundant = 0; + foreach my $acc (@accs) { + if ($acc_counter{$acc} > 1) { + $num_redundant++; + } + } + print join("\t", $file, $num_total_accs - $num_redundant, $num_redundant) . "\n"; + } + + + print "\nHistogram of reconstructions across threads.\n"; + my @counts = sort {$a<=>$b} keys %count_counter; + foreach my $count (@counts) { + print "$count\t$count_counter{$count}\n"; + } + print "\n"; # spacer + + +} + + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/examine_strand_specificity.pl b/99.scripts/trinity_utils/util/misc/examine_strand_specificity.pl new file mode 100644 index 0000000..eaf18eb --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/examine_strand_specificity.pl @@ -0,0 +1,92 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use SAM_reader; +use SAM_entry; +use Process_cmd; + +my $usage = "\n\n\tusage: $0 transcript_aligned.bam [out_prefix='ss_analysis']\n\n"; + +my $bam_file = $ARGV[0] or die $usage; +my $out_prefix = $ARGV[1] || "ss_analysis"; + +main: { + + my %transcript_to_orients; + + my $sam_reader = new SAM_reader($bam_file); + print STDERR "-parsing file: $bam_file\n"; + while (my $sam_entry = $sam_reader->get_next()) { + + my $trans_name = $sam_entry->get_scaffold_name(); + my $orient = $sam_entry->get_query_strand(); + + + if ($sam_entry->is_paired()) { + + unless ($sam_entry->is_proper_pair() && $sam_entry->is_first_in_pair()) { + next; + } + } + + $transcript_to_orients{$trans_name}->{$orient}++; + + } + print STDERR "-done parsing file, examining orientations of reads.\n"; + + ## sum them up. + my @transcripts = keys %transcript_to_orients; + foreach my $transcript (@transcripts) { + + my $orient_plus = $transcript_to_orients{$transcript}->{'+'} || 0; + my $orient_minus = $transcript_to_orients{$transcript}->{'-'} || 0; + + + my $total_reads = $orient_plus + $orient_minus; + + $transcript_to_orients{$transcript}->{'transcript'} = $transcript; + $transcript_to_orients{$transcript}->{'total_reads'} = $total_reads; + } + + #### + my @structs = values %transcript_to_orients; + @structs = reverse sort {$a->{total_reads}<=>$b->{total_reads}} @structs; + + # header + open (my $ofh, ">$out_prefix.dat") or die "Error, cannot write to $out_prefix.dat"; + + print $ofh join("\t", "#transcript", "plus_strand_1stReads", "minus_strand_1stReads", "total_reads", "diff_ratio") . "\n"; + + foreach my $struct (@structs) { + + my $struct = shift @structs; + + + my ($plus, $minus, $total_reads, $transcript) = ($struct->{'+'}, + $struct->{'-'}, + $struct->{'total_reads'}, + $struct->{'transcript'}); + unless ($plus) { $plus = 0; } + unless ($minus) { $minus = 0; } + + my $diff_proportion = sprintf("%.3f", ($plus - $minus) / $total_reads); + + print $ofh join("\t", $transcript, $plus, $minus, $total_reads, $diff_proportion) . "\n"; + + } + close $ofh; + + + ## plot it. + my $cmd = "$FindBin::Bin/plot_strand_specificity_dist_by_quantile.Rscript $out_prefix.dat"; + &process_cmd($cmd); + + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/examine_weldmer_halves.pl b/99.scripts/trinity_utils/util/misc/examine_weldmer_halves.pl new file mode 100644 index 0000000..7afaaae --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/examine_weldmer_halves.pl @@ -0,0 +1,108 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "\n\nusage: $0 weldmer\n\n"; + +my $weldmer = $ARGV[0] or die $usage; + +main: { + + if (-f $weldmer) { + open (my $fh, $weldmer) or die $!; + while (<$fh>) { + my $weldmer = $_; + chomp $weldmer; + &check_weldmer($weldmer); + } + close $fh; + } + else { + &check_weldmer($weldmer); + } + + exit(0); +} + + +#### +sub check_weldmer { + my ($weldmer) = @_; + + my $weldmer_len = length($weldmer); + + my $left_substr = substr($weldmer, 0, int($weldmer_len/2)); + my $right_substr = substr($weldmer, int($weldmer_len/2)); + + # &check_per_id($weldmer); + &check_per_id($left_substr); + &check_per_id($right_substr); + + + return; + +} + + +#### +sub check_per_id { + my ($weldmer) = @_; + + my $weldmer_len = length($weldmer); + + my $half_len = int($weldmer_len/2); + #print "Half_len of " . length($weldmer_len) . " = $half_len\n"; + + my @chars = split(//, $weldmer); + + my $max_ratio = 0; + + my $best_left = ""; + my $best_right = ""; + + for (my $i = 0; $i < $half_len; $i++) { + + for (my $j = $i + 1; $j <= $half_len; $j++) { + + my $ref_pos = $i; + my $other_pos = $j; + + my $count_same = 0; + my $count_chars = 0; + + my $left_sub = ""; + my $right_sub = ""; + + + + while ($other_pos <= $j + $half_len - 1) { + + $count_chars++; + if ($chars[$ref_pos] eq $chars[$other_pos]) { + $count_same++; + } + + $left_sub .= $chars[$ref_pos]; + $right_sub .= $chars[$other_pos]; + + $ref_pos++; + $other_pos++; + + } + + my $ratio = $count_same/$count_chars; + if ($ratio > $max_ratio) { + $max_ratio = $ratio; + $best_left = $left_sub; + $best_right = $right_sub; + } + + } + } + + $max_ratio = sprintf("%.3f", $max_ratio); + print "$weldmer\t$best_left\t$best_right\t$max_ratio\n"; + + return; +} diff --git a/99.scripts/trinity_utils/util/misc/extract_fastQ_pairings.pl b/99.scripts/trinity_utils/util/misc/extract_fastQ_pairings.pl new file mode 100644 index 0000000..623c0b2 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/extract_fastQ_pairings.pl @@ -0,0 +1,223 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Data::Dumper; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fastq_reader; + +my $DEBUG = 0; + + +my $usage = "usage: $0 left.fq right.fq\n\n"; + +my $left_fq = $ARGV[0] or die $usage; +my $right_fq = $ARGV[1] or die $usage; + +open (my $left_PP_ofh, ">$left_fq.P.fq") or die $!; +open (my $left_UP_ofh, ">$left_fq.U.fq") or die $!; + +open (my $right_PP_ofh, ">$right_fq.P.fq") or die $!; +open (my $right_UP_ofh, ">$right_fq.U.fq") or die $!; + +my $ok_counter = 0; +my $left_orphan_counter = 0; +my $right_orphan_counter = 0; + +main: { + + my $left_fq_reader = new Fastq_reader($left_fq); + my $right_fq_reader = new Fastq_reader($right_fq); + + + + my ($left_fq_record, $right_fq_record); + + my @left_entries; + my @right_entries; + + my %core_counter; + + do { + + + my $num_left_stored = scalar(@left_entries); + my $num_right_stored = scalar(@right_entries); + + + if ($DEBUG) { + + my %seen; + + foreach my $left_entry (@left_entries) { + print STDERR "L " . $left_entry->get_full_read_name() . "\n" if $DEBUG; + my $core_acc = $left_entry->get_core_read_name(); + $seen{$core_acc}++; + } + print STDERR "\n" if $DEBUG; + my $found_hit = 0; + foreach my $right_entry (@right_entries) { + print STDERR "R " . $right_entry->get_full_read_name() . "\n" if $DEBUG; + my $core_acc = $right_entry->get_core_read_name(); + my $count = ++$seen{$core_acc}; + if ($count == 2) { + print STDERR " ***** \n" if $DEBUG; + $found_hit++; + } + } + print STDERR "\n\n" if $DEBUG; + + if ($found_hit) { + die " reads must be jumbled"; + } + } + + my $MAX_ORPHAN_STORE = 100; + #if ($num_left_stored > $MAX_ORPHAN_STORE && $num_right_stored > $MAX_ORPHAN_STORE) { die; } + + + if ($ok_counter % 1000 == 0) { + print STDERR "\r[$ok_counter pairs_written, left_orphans_written: $left_orphan_counter, right_orphans_written: $right_orphan_counter] "; + print STDERR "[Left cache:$num_left_stored, Right cache:$num_right_stored] "; + } + + $left_fq_record = $left_fq_reader->next(); + push (@left_entries, $left_fq_record); + + my $left_core_acc = $left_fq_record->get_core_read_name(); + my $count = ++$core_counter{$left_core_acc}; + + if ($count == 2) { + &dump_pairs($left_core_acc, \@left_entries, \@right_entries, \%core_counter); + } + + $right_fq_record = $right_fq_reader->next(); + push (@right_entries, $right_fq_record); + + my $right_core_acc = $right_fq_record->get_core_read_name(); + $count = ++$core_counter{$right_core_acc}; + + if ($count == 2) { + &dump_pairs($right_core_acc, \@left_entries, \@right_entries, \%core_counter); + } + + + #print STDERR Dumper(\%core_counter); + + + + + } while ($left_fq_record && $right_fq_record); + + while (@left_entries) { + + $left_fq_record = shift @left_entries; + + print $left_UP_ofh $left_fq_record->get_fastq_record(); + + $left_orphan_counter++; + } + + while (@right_entries) { + + $right_fq_record = shift @right_entries; + + print $right_UP_ofh $right_fq_record->get_fastq_record(); + + $right_orphan_counter++; + + } + + print STDERR "\r[$ok_counter pairs_written, left_orphans_written: $left_orphan_counter, right_orphans_written: $right_orphan_counter]\n\nDone.\n\n"; + + exit(0); +} + +#### +sub dump_pairs { + my ($acc, $left_entries_aref, $right_entries_aref, $core_counter_href) = @_; + + if ($left_entries_aref->[ $#$left_entries_aref ]->get_core_read_name() eq $acc) { + + my $record = pop @$left_entries_aref; + print $left_PP_ofh $record->get_fastq_record(); + + ## write earlier stored records as unpaired entries + while ($record = shift @$left_entries_aref) { + + my $core_acc = $record->get_core_read_name(); + delete $core_counter_href->{$core_acc}; + + print $left_UP_ofh $record->get_fastq_record(); + $left_orphan_counter++; + } + + # process right records + while ($record = shift @$right_entries_aref) { + + my $core_acc = $record->get_core_read_name(); + if ($core_acc eq $acc) { + print $right_PP_ofh $record->get_fastq_record(); + last; # retain any remaining entries on the stack + + } + else { + + my $core_acc = $record->get_core_read_name(); + delete $core_counter_href->{$core_acc}; + print $right_UP_ofh $record->get_fastq_record(); + + $right_orphan_counter++; + } + + } + } + elsif ($right_entries_aref->[ $#$right_entries_aref ]->get_core_read_name() eq $acc) { + + my $record = pop @$right_entries_aref; + print $right_PP_ofh $record->get_fastq_record(); + + ## write earlier stored records as unpaired entries + while ($record = shift @$right_entries_aref) { + + my $core_acc = $record->get_core_read_name(); + delete $core_counter_href->{$core_acc}; + + print $right_UP_ofh $record->get_fastq_record(); + + $right_orphan_counter++; + } + + # process left records + while ($record = shift @$left_entries_aref) { + + my $core_acc = $record->get_core_read_name(); + if ($core_acc eq $acc) { + print $left_PP_ofh $record->get_fastq_record(); + last; # retain any remaining entries on the stack + + } + else { + + my $core_acc = $record->get_core_read_name(); + delete $core_counter_href->{$core_acc}; + print $left_UP_ofh $record->get_fastq_record(); + + $left_orphan_counter++; + } + + } + } + + delete $core_counter_href->{$acc}; + + print STDERR "\n\nOK: $acc\n" if $DEBUG; + + $ok_counter += 2; + + return; +} + diff --git a/99.scripts/trinity_utils/util/misc/fan_out_fasta_seqs_to_indiv_files.pl b/99.scripts/trinity_utils/util/misc/fan_out_fasta_seqs_to_indiv_files.pl new file mode 100644 index 0000000..a39dcdf --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fan_out_fasta_seqs_to_indiv_files.pl @@ -0,0 +1,110 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; + +use lib ($ENV{EUK_MODULES}); +use Fasta_reader; + + +my $usage = "\n\n\tusage: $0 fasta_file min_seq_len (byGene|byTrans) [numPerDir=100]\n\nnote: byGene restricts to multi-iso genes\n\n"; + +my $fasta_file = $ARGV[0] or die $usage; +my $min_seq_length = $ARGV[1] or die $usage; +my $by_gene_or_trans = $ARGV[2] or die $usage; +my $num_per_dir = $ARGV[3] || 100; + +unless ($by_gene_or_trans =~ /^(byGene|byTrans)$/) { die $usage; } + +main: { + + my $fasta_reader = new Fasta_reader($fasta_file); + my %seqs = $fasta_reader->retrieve_all_seqs_hash(); + + my $by_gene_flag = ($by_gene_or_trans eq 'byGene'); + + %seqs = &reorganize_seqs(\%seqs, $by_gene_flag); + + my $outdir = "$by_gene_or_trans.dir"; + + mkdir($outdir) or die "Error, cannot mkdir $outdir"; + + open (my $ofh_file_listing, ">$outdir.listing") or die "Error, cannot write to $outdir.listing"; + + my $counter = 0; + foreach my $acc (keys %seqs) { + my $bindir = "$outdir/bin." . int($counter/$num_per_dir); + if (! -d $bindir) { + mkdir ($bindir) or die "Error, cannot mkdir $bindir"; + } + + my $acc_for_filename = $acc; + $acc_for_filename =~ s/\W/_/g; + + my @seq_entries = @{$seqs{$acc}}; + + if ($by_gene_flag && scalar(@seq_entries) == 1) { + ## only focusing on multi-iso genes. + next; + } + + + my $filename = "$bindir/$acc_for_filename.ref.fa"; + open (my $ofh, ">$filename") or die "Error, cannot write to $filename"; + + + foreach my $seq_entry (@seq_entries) { + my ($acc, $seq) = @$seq_entry; + print $ofh ">$acc\n$seq\n"; + } + close $ofh; + + print $ofh_file_listing "$filename\n"; + print STDERR "// wrote $filename\n"; + + $counter++; + } + + + print STDERR "\n\nDone.\n\n"; + + exit(0); + + +} + + + +#### +sub reorganize_seqs { + my ($fasta_seqs_href, $by_gene_flag) = @_; + + my %reorg_fasta; + + foreach my $acc (sort keys %$fasta_seqs_href) { + + my $seq = uc $fasta_seqs_href->{$acc}; + + $seq =~ s/N//g; # no N characters allowed. + + unless (length($seq) >= $min_seq_length) { next; } + + my $key = $acc; + if ($by_gene_flag) { + my ($trans, $gene) = split(/;/, $acc); + unless ($gene) { + confess "Error, no gene ID extracted from $acc "; + } + $key = $gene; + } + + + push (@{$reorg_fasta{$key}}, [$acc, $seq]); + } + + return(%reorg_fasta); + +} + + diff --git a/99.scripts/trinity_utils/util/misc/fastQ_append_acc.pl b/99.scripts/trinity_utils/util/misc/fastQ_append_acc.pl new file mode 100644 index 0000000..08d6fd7 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fastQ_append_acc.pl @@ -0,0 +1,37 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 file.fastq append_no\n\n"; + +my $fastq_file = $ARGV[0] or die $usage; +my $append_no = $ARGV[1] or die $usage; + + +my @lines; + +my $counter = 0; +open (my $fh, $fastq_file) or die "Error, cannot open file $fastq_file\n"; +while (<$fh>) { + chomp; + $counter++; + push (@lines, $_); + + if ($counter % 4 == 0) { + if ($lines[0] !~ /^\@/) { + die "Error, fastq record doesn't start with a header line as expected: " . join("\n", @lines); + } + + my @x = split(/\s+/, $lines[0]); + $x[0] .= "/$append_no"; + $lines[0] = join(" ", @x); + print join("\n", @lines) . "\n"; + @lines = (); + } + +} +close $fh; + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/fastQ_rand_subset.SE.reservoir_sampling_reqiures_high_mem.pl b/99.scripts/trinity_utils/util/misc/fastQ_rand_subset.SE.reservoir_sampling_reqiures_high_mem.pl new file mode 100644 index 0000000..48fcdcd --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fastQ_rand_subset.SE.reservoir_sampling_reqiures_high_mem.pl @@ -0,0 +1,95 @@ +#!/usr/bin/env perl + +## util/fastQ_rand_subset.pl +## Purpose: Extracts out a specific number of random reads from an +## input file, making sure that left and right ends of the selected +## reads are paired +## Usage: $0 left.fq right.fq num_entries +## +## This code implements reservoir sampling, which produces an unbiased +## selection regardless of the number of entries (assuming an +## appropriately uniform random distribution function). + +## See: +## * https://en.wikipedia.org/wiki/Reservoir_sampling +## +## The current implementation requires the selected entries to be +## stored in memory, but passes over the input file(s) only once. If +## the array loading / reading is too slow or memory intensive, this +## could be implemented in a two-pass fashion by first getting +## indexes, and then getting the reads + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fastq_reader; +use File::Basename; + +my $usage = "usage: $0 single.fq num_entries\n\n"; + +my $left_fq = $ARGV[0] or die $usage; +my $num_entries = $ARGV[1] or die $usage; + +my @selected_left_entries = (); +my @selected_right_entries = (); + +main: { + + print STDERR "Selecting $num_entries entries..."; + + my $left_fq_reader = new Fastq_reader($left_fq) or die("unable to open $left_fq for input"); + + my $num_M_entries = $num_entries/1e6; + $num_M_entries .= "M"; + my $base_left_fq = basename($left_fq); + + open (my $left_ofh, ">$base_left_fq.$num_M_entries.fq") or die $!; + + srand(); + + my $num_skipped = 0; + my $num_output_entries = 0; + my $num_entries_read = 0; + + my $left_entry = 0; + my $right_entry = 0; + + while ($left_entry = $left_fq_reader->next()) { + + + $num_entries_read++; + + if($num_entries_read <= $num_entries){ + # Populate reservoir with entries up to num_entries + push(@selected_left_entries, $left_entry); + + } else { + # Randomly replace elements in the reservoir + # with a decreasing probability + my $swapPos = int(rand($num_entries_read)); + + if($swapPos < $num_entries){ + $selected_left_entries[$swapPos] = $left_entry; + } + } + } + + if ($num_entries > $num_entries_read) { + die "Error, num_entries $num_entries > total records available: $num_entries_read "; + } + + # print selected entries to their respective files + foreach my $entry (@selected_left_entries){ + print $left_ofh $entry->get_fastq_record(); + } + + print STDERR " done.\n"; + + close $left_ofh; + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/fastQ_rand_subset.pl b/99.scripts/trinity_utils/util/misc/fastQ_rand_subset.pl new file mode 100644 index 0000000..8b957c9 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fastQ_rand_subset.pl @@ -0,0 +1,99 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fastq_reader; +use File::Basename; + +my $usage = "usage: $0 left.fq right.fq num_entries\n\n"; + +my $left_fq = $ARGV[0] or die $usage; +my $right_fq = $ARGV[1] or die $usage; +my $num_to_sample = $ARGV[2] or die $usage; + +srand(); + +main: { + + ## do reservoir sampling on indices instead of the fastq records themselves, then just extract the selected ones. + + my $num_fq_entries = &get_num_fq_entries($left_fq); + my @selected_entries = (1..$num_to_sample); + + for (my $i = $num_to_sample + 1; $i <= $num_fq_entries; $i++) { + + my $rand_pos = rand($i); + if ($rand_pos < $num_to_sample) { + $selected_entries[$rand_pos] = $i; + } + } + + my %indices_want = map { + $_ => 1 } @selected_entries; + @selected_entries = (); # free + + + my $left_fq_reader = new Fastq_reader($left_fq) or die("unable to open $left_fq for input"); + my $right_fq_reader = new Fastq_reader($right_fq) or die("unable to open $right_fq for input");; + + my $num_M_entries = $num_to_sample/1e6; + $num_M_entries .= "M"; + my $base_left_fq = basename($left_fq); + my $base_right_fq = basename($right_fq); + + open (my $left_ofh, ">$base_left_fq.$num_M_entries.fq") or die $!; + open (my $right_ofh, ">$base_right_fq.$num_M_entries.fq") or die $!; + + my $fq_index = 0; + while (my $left_entry = $left_fq_reader->next()) { + my $right_entry = $right_fq_reader->next(); + + unless ($left_entry && $right_entry) { + die "Error, didn't retrieve both left and right entries from file ($left_entry, $right_entry) "; + } + unless ($left_entry->get_core_read_name() eq $right_entry->get_core_read_name()) { + die "Error, core read names don't match: " + . "Left: " . $left_entry->get_core_read_name() . "\n" + . "Right: " . $right_entry->get_core_read_name() . "\n"; + } + + $fq_index++; + + if ($indices_want{$fq_index}) { + print $left_ofh $left_entry->get_fastq_record(); + print $right_ofh $right_entry->get_fastq_record(); + } + } + + print STDERR " done.\n"; + + close $left_ofh; + close $right_ofh; + + exit(0); +} + +#### +sub get_num_fq_entries { + my ($fastq_file) = @_; + + unless (-s $fastq_file) { + die "Error, cannot locate file $fastq_file"; + } + + my $linecount; + if ($fastq_file =~ /\.gz$/) { + $linecount = `gunzip -c $fastq_file | wc -l`; + } + else { + $linecount = `cat $fastq_file | wc -l`; + } + chomp $linecount; + + my $num_records = $linecount / 4; + + return($num_records); +} + diff --git a/99.scripts/trinity_utils/util/misc/fastQ_rand_subset.reservoir_sampling_reqiures_high_mem.pl b/99.scripts/trinity_utils/util/misc/fastQ_rand_subset.reservoir_sampling_reqiures_high_mem.pl new file mode 100644 index 0000000..eedc324 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fastQ_rand_subset.reservoir_sampling_reqiures_high_mem.pl @@ -0,0 +1,113 @@ +#!/usr/bin/env perl + +## util/fastQ_rand_subset.pl +## Purpose: Extracts out a specific number of random reads from an +## input file, making sure that left and right ends of the selected +## reads are paired +## Usage: $0 left.fq right.fq num_entries +## +## This code implements reservoir sampling, which produces an unbiased +## selection regardless of the number of entries (assuming an +## appropriately uniform random distribution function). + +## See: +## * https://en.wikipedia.org/wiki/Reservoir_sampling +## +## The current implementation requires the selected entries to be +## stored in memory, but passes over the input file(s) only once. If +## the array loading / reading is too slow or memory intensive, this +## could be implemented in a two-pass fashion by first getting +## indexes, and then getting the reads + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fastq_reader; +use File::Basename; + +my $usage = "usage: $0 left.fq right.fq num_entries\n\n"; + +my $left_fq = $ARGV[0] or die $usage; +my $right_fq = $ARGV[1] or die $usage; +my $num_entries = $ARGV[2] or die $usage; + +my @selected_left_entries = (); +my @selected_right_entries = (); + +main: { + + print STDERR "Selecting $num_entries entries..."; + + my $left_fq_reader = new Fastq_reader($left_fq) or die("unable to open $left_fq for input"); + my $right_fq_reader = new Fastq_reader($right_fq) or die("unable to open $right_fq for input");; + + my $num_M_entries = $num_entries/1e6; + $num_M_entries .= "M"; + my $base_left_fq = basename($left_fq); + my $base_right_fq = basename($right_fq); + + open (my $left_ofh, ">$base_left_fq.$num_M_entries.fq") or die $!; + open (my $right_ofh, ">$base_right_fq.$num_M_entries.fq") or die $!; + + srand(); + + my $num_skipped = 0; + my $num_output_entries = 0; + my $num_entries_read = 0; + + my $left_entry = 0; + my $right_entry = 0; + + while ($left_entry = $left_fq_reader->next()) { + $right_entry = $right_fq_reader->next(); + + unless ($left_entry && $right_entry) { + die "Error, didn't retrieve both left and right entries from file ($left_entry, $right_entry) "; + } + unless ($left_entry->get_core_read_name() eq $right_entry->get_core_read_name()) { + die "Error, core read names don't match: " + . "Left: " . $left_entry->get_core_read_name() . "\n" + . "Right: " . $right_entry->get_core_read_name() . "\n"; + } + + $num_entries_read++; + + if($num_entries_read <= $num_entries){ + # Populate reservoir with entries up to num_entries + push(@selected_left_entries, $left_entry); + push(@selected_right_entries, $right_entry); + } else { + # Randomly replace elements in the reservoir + # with a decreasing probability + my $swapPos = int(rand($num_entries_read)); + + if($swapPos < $num_entries){ + $selected_left_entries[$swapPos] = $left_entry; + $selected_right_entries[$swapPos] = $right_entry; + } + } + } + + if ($num_entries > $num_entries_read) { + die "Error, num_entries $num_entries > total records available: $num_entries_read "; + } + + # print selected entries to their respective files + foreach my $entry (@selected_left_entries){ + print $left_ofh $entry->get_fastq_record(); + } + foreach my $entry (@selected_right_entries){ + print $right_ofh $entry->get_fastq_record(); + } + + print STDERR " done.\n"; + + close $left_ofh; + close $right_ofh; + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/fastQ_top_N_records.pl b/99.scripts/trinity_utils/util/misc/fastQ_top_N_records.pl new file mode 100644 index 0000000..8a27735 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fastQ_top_N_records.pl @@ -0,0 +1,35 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 file.fq TopEntriesCount [append]\n\n"; + +my $fq_file = $ARGV[0] or die $usage; +my $top_entries_count = $ARGV[1] or die $usage; +my $append = $ARGV[2]; + +my $count = 0; +open (my $fh, $fq_file) or die "Error, cannot open file $fq_file"; +while (my $line1 = <$fh>) { + my $line2 = <$fh>; + my $line3 = <$fh>; + my $line4 = <$fh>; + + chomp ($line1, $line2, $line3, $line4); + + $count++; + + if ($top_entries_count > 0 && $count > $top_entries_count) { + last; + } + + if ($append) { + $line1 .= "/" . $append; + } + + print join("\n", $line1, $line2, $line3, $line4) . "\n"; +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/fasta_file_reformatter.pl b/99.scripts/trinity_utils/util/misc/fasta_file_reformatter.pl new file mode 100644 index 0000000..8f0be71 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fasta_file_reformatter.pl @@ -0,0 +1,29 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + +my $usage = "usage: $0 fasta\n"; + +my $fasta_file = $ARGV[0] or die $usage; + +my $fasta_reader = new Fasta_reader($fasta_file); + +while (my $seq_obj = $fasta_reader->next()) { + + my $header = $seq_obj->get_header(); + my $seq = $seq_obj->get_sequence(); + + $seq =~ s/(\S{60})/$1\n/g; + + chomp $seq; + print ">$header\n$seq\n"; + +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/fasta_filter_by_min_length.pl b/99.scripts/trinity_utils/util/misc/fasta_filter_by_min_length.pl new file mode 100644 index 0000000..167ccab --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fasta_filter_by_min_length.pl @@ -0,0 +1,30 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + + +my $usage = "usage: $0 fastaFile min_length\n\n"; + +my $fasta_file = $ARGV[0] or die $usage; +my $min_length = $ARGV[1] or die $usage; + +my $fasta_reader = new Fasta_reader($fasta_file); + +while (my $seqobj = $fasta_reader->next()) { + my $fasta_entry = $seqobj->get_FASTA_format(); + my $sequence = $seqobj->get_sequence(); + unless (length($sequence) >= $min_length) { + next; + } + + print $fasta_entry; +} + + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/fasta_remove_duplicates.pl b/99.scripts/trinity_utils/util/misc/fasta_remove_duplicates.pl new file mode 100644 index 0000000..d28fe01 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fasta_remove_duplicates.pl @@ -0,0 +1,37 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use lib ($ENV{EUK_MODULES}); +use Fasta_reader; + + +my $usage = "usage: $0 seqs.fasta\n\n"; + +my $file = $ARGV[0] or die $usage; + +my %seq_to_header; + +my $fasta_reader = new Fasta_reader($file); +while (my $seq_obj = $fasta_reader->next()) { + + my $sequence = $seq_obj->get_sequence(); + my $header = $seq_obj->get_header(); + + if (exists $seq_to_header{$sequence}) { + $seq_to_header{$sequence} .= "\t$header"; + } + else { + $seq_to_header{$sequence} = $header; + } +} + +foreach my $sequence (keys %seq_to_header) { + my $header = $seq_to_header{$sequence}; + + print ">$header\n$sequence\n"; +} + +exit(0); + + diff --git a/99.scripts/trinity_utils/util/misc/fasta_seq_length.pl b/99.scripts/trinity_utils/util/misc/fasta_seq_length.pl new file mode 100644 index 0000000..26a70b9 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fasta_seq_length.pl @@ -0,0 +1,25 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; + +use lib ("$FindBin::Bin/../../PerlLib"); +use Fasta_reader; + +my $usage = "usage: $0 fastaFile\n\n"; + +my $file = $ARGV[0] or die $usage; + +my $fasta_reader = new Fasta_reader($file); + +print join("\t", "#fasta_entry", "length") . "\n"; +while (my $seq_obj = $fasta_reader->next()) { + my $sequence = $seq_obj->get_sequence(); + my $accession = $seq_obj->get_accession(); + + print join("\t", $accession, length($sequence)) . "\n"; +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/fasta_to_cmd_generator.pl b/99.scripts/trinity_utils/util/misc/fasta_to_cmd_generator.pl new file mode 100644 index 0000000..53f0dea --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fasta_to_cmd_generator.pl @@ -0,0 +1,163 @@ +#!/usr/bin/env perl + +use strict; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; +use Getopt::Std; +use strict; +use Carp; +use Cwd; + +use List::Util qw (shuffle); + +our ($opt_d, $opt_q, $opt_s, $opt_Q, $opt_b, $opt_p, $opt_O, $opt_h, $opt_c, $opt_B, $opt_X, $opt_M, $opt_S, $opt_o); + +&getopts ('dq:s:bp:O:hbc:B:Q:XM:S:o:'); + +my $usage = <<_EOH_; + +############################# Options ############################### +# +# Required: +# -q query multiFastaFile (full or relative path) +# -p program command line template: eg. "/path/to/prog [opts] __QUERY_FILE__ [other opts]" +# -o outdir +# +# Optional: +# -S number of fasta seqs per job submission (default: 1) +# -B bin size (input seqs per directory) (default 5000) +# +###################### Process Args and Options ##################### + +_EOH_ + + + ; + + +if ($opt_h) { + die $usage; +} + +my $CMDS_ONLY = $opt_X; + +my $bin_size = $opt_B || 5000; + +our $DEBUG = $opt_d; + +my $num_seqs_per_job = $opt_S || 1; + +unless ($opt_q && $opt_p && $opt_o) { + die $usage; +} + +my $queryFile = $opt_q; +unless ($queryFile =~ /^\//) { + $queryFile = cwd() . "/$queryFile"; +} + +my $program_cmd_template = $opt_p; +unless ($program_cmd_template =~ /__QUERY_FILE__/) { + die "Error, program cmd template must include '__QUERY_FILE__' placeholder in the command"; +} + + +my $out_dir = $opt_o; + +## Create files to search + +my $fastaReader = new Fasta_reader($queryFile); + +my @searchFileList; + +my $count = 0; +my $current_bin = 1; + +mkdir $out_dir or die "Error, cannot mkdir $out_dir"; + +my $bindir = "$out_dir/grp_" . sprintf ("%04d", $current_bin); +mkdir ($bindir) or die "Error, cannot mkdir $bindir"; + + + +while (my $fastaSet = &get_next_fasta_entries($fastaReader, $num_seqs_per_job) ) { + + $count++; + + my $filename = "$bindir/$count.fa"; + + push (@searchFileList, $filename); + + open (TMP, ">$filename") or die "Can't create file ($filename)\n"; + print TMP $fastaSet; + close TMP; + chmod (0666, $filename); + + if ($count % $bin_size == 0) { + # make a new bin: + $current_bin++; + $bindir = "$out_dir/grp_" . sprintf ("%04d", $current_bin); + mkdir ($bindir) or die "Error, cannot mkdir $bindir"; + } +} + +my $numFiles = @searchFileList; + +my $curr_dir = cwd; + +if ($numFiles) { + + my @cmds; + ## formulate blast commands: + foreach my $searchFile (@searchFileList) { + $searchFile = "$curr_dir/$searchFile"; + + my $cmd = $program_cmd_template; + $cmd =~ s/__QUERY_FILE__/$searchFile/g; + + + $cmd .= " > $searchFile.OUT "; + + $cmd .= " 2>$searchFile.ERR"; + + push (@cmds, $cmd); + } + + + @cmds = shuffle(@cmds); + + open (my $fh, ">$out_dir.cmds.list") or die $!; + foreach my $cmd (@cmds) { + print $fh "$cmd\n"; + } + close $fh; + + + print STDERR "\n\nCommands written to: $out_dir.cmds.list\n\n"; + +} + + +exit(0); + + +#### +sub get_next_fasta_entries { + my ($fastaReader, $num_seqs) = @_; + + + my $fasta_entries_txt = ""; + + for (1..$num_seqs) { + my $seq_obj = $fastaReader->next(); + unless ($seq_obj) { + last; + } + + my $entry_txt = $seq_obj->get_FASTA_format(); + $fasta_entries_txt .= $entry_txt; + } + + return($fasta_entries_txt); +} diff --git a/99.scripts/trinity_utils/util/misc/fasta_to_tab.pl b/99.scripts/trinity_utils/util/misc/fasta_to_tab.pl new file mode 100644 index 0000000..26e1f01 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fasta_to_tab.pl @@ -0,0 +1,37 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use Fasta_reader; + +my $usage = "usage: $0 [multiFastaFile]\n\n"; + +if (defined($ARGV[0]) && ($ARGV[0] eq "-h" || $ARGV[0] eq "--help")) { + die $usage; +} + + +my $input = $ARGV[0] || *STDIN{IO}; + +unless (-f $input || ref $input eq 'IO::Handle') { + die "Error, input not established."; +} + +main: { + + my $fasta_reader = new Fasta_reader($input); + + while (my $seq_obj = $fasta_reader->next()) { + my $sequence = $seq_obj->get_sequence(); + my $header = $seq_obj->get_header(); + + print "$header\t$sequence\n"; + } + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/fasta_write_sense_n_anti.pl b/99.scripts/trinity_utils/util/misc/fasta_write_sense_n_anti.pl new file mode 100644 index 0000000..a44b88a --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fasta_write_sense_n_anti.pl @@ -0,0 +1,37 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; +use Nuc_translator; + +my $usage = "\n\nusage: $0 transcripts.fa\n\n"; + +my $transcripts_file = $ARGV[0] or die $usage; + + +main: { + + my $fasta_reader = new Fasta_reader($transcripts_file); + + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + + my $seq = $seq_obj->get_sequence(); + + print ">$acc\n$seq\n"; + + my $revc_seq = &reverse_complement($seq); + + print ">$acc-ANTI\n$revc_seq\n"; + + } + + + exit(0); + +} diff --git a/99.scripts/trinity_utils/util/misc/fastq_cleaner.pl b/99.scripts/trinity_utils/util/misc/fastq_cleaner.pl new file mode 100644 index 0000000..1e9734a --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fastq_cleaner.pl @@ -0,0 +1,77 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 inputFile out.cleanReads out.malformedReads\n\n"; + +my $inputFile = $ARGV[0] or die $usage; +my $cleanReads = $ARGV[1] or die $usage; +my $malformedReads = $ARGV[2] or die $usage; + + +open (my $ofh_clean, ">$cleanReads") or die "Error, can't write to $cleanReads"; +open (my $ofh_malformed, ">$malformedReads") or die "Error, cannot write to $malformedReads"; + +open (my $fh, $inputFile) or die "Error, cannot open $inputFile"; + +my $counter = 0; +my $num_clean = 0; +my $num_dirty = 0; + +my @rec; + +my $line = <$fh>; + +while ($line) { + + if ($line =~ /^\@/) { + $counter++; + + print STDERR "\r[$counter] [$num_clean clean] [$num_dirty dirty] " if ($counter % 10000 == 0); + + push (@rec, $line); + + $line = <$fh>; + for (1..3) { + push (@rec, $line); + $line = <$fh>; + } + + my $record_text = join("", @rec); + + my $header = shift @rec; + my $seq = shift @rec; + my $qual_header = shift @rec; + my $qual_line = shift @rec; + + chomp $header; + chomp $seq if $seq; + chomp $qual_header if $qual_header; + chomp $qual_line if $qual_line; + + if ($header && $seq && $qual_header && $qual_line && + $qual_header =~ /^\+/ && length($seq) == length($qual_line)) { + + # can do some more checks here if needed to be sure that the lines are formatted as expected. + print $ofh_clean join("\n", $header, $seq, $qual_header, $qual_line) . "\n"; + $num_clean++; + } + else { + print $ofh_malformed $record_text; + $num_dirty++; + } + @rec = (); + } else { + $line = <$fh>; + } + +} + +exit(0); + + + + + + diff --git a/99.scripts/trinity_utils/util/misc/fastq_interleave_pairs.pl b/99.scripts/trinity_utils/util/misc/fastq_interleave_pairs.pl new file mode 100644 index 0000000..ec749ab --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fastq_interleave_pairs.pl @@ -0,0 +1,57 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use Fastq_reader; + +my $usage = "\n\nusage: $0 left.fq right.fq\n\n"; + +my $left_fq = $ARGV[0] or die $usage; +my $right_fq = $ARGV[1] or die $usage; + +main: { + + my $left_fastq_reader = new Fastq_reader($left_fq); + my $right_fastq_reader = new Fastq_reader($right_fq); + + while (1) { + my $left_fq = $left_fastq_reader->next(); + if ($left_fq) { + print $left_fq->get_fastq_record(); + } + + + my $right_fq = $right_fastq_reader->next(); + if ($right_fq) { + print $right_fq->get_fastq_record(); + } + + if ($left_fq && $right_fq) { + my $left_core_read_name = $left_fq->get_core_read_name(); + my $right_core_read_name = $right_fq->get_core_read_name(); + + if ($left_core_read_name ne $right_core_read_name) { + die "Error, read names are out of synch: $left_core_read_name vs. $right_core_read_name "; + } + } + + else { + last; + } + + } + + if ($left_fastq_reader->next() || $right_fastq_reader->next()) { + die "Error, unequal number of fastq records in left and right fq files."; + } + + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/fastq_merge_sorted_tab_lists.pl b/99.scripts/trinity_utils/util/misc/fastq_merge_sorted_tab_lists.pl new file mode 100644 index 0000000..7ca1b42 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fastq_merge_sorted_tab_lists.pl @@ -0,0 +1,91 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + + +my $usage = "usage: left.tab.sort right.tab.sort\n\n"; + +my $left_tab = $ARGV[0] or die $usage; +my $right_tab = $ARGV[1] or die $usage; + + +open (my $left_fh, $left_tab) or die $!; +open (my $right_fh, $right_tab) or die $!; + +my $left_entry = <$left_fh>; +my $right_entry = <$right_fh>; + +open (my $left_ofh, ">fixed.$$.left.fq") or die $!; +open (my $right_ofh, ">fixed.$$.right.fq") or die $!; + +open (my $left_broken_ofh, ">broken.$$.left.fq") or die $!; +open (my $right_broken_ofh, ">broken.$$.right.fq") or die $!; + +my $counter=0; + +while (1) { + + $counter++; + if ($counter % 1000000 == 0) { + print STDERR "\r[$counter] "; + } + + unless ($left_entry && $right_entry) { + last; + } + + chomp $left_entry; + chomp $right_entry; + + my ($left_acc, $left_seq, $left_qual) = split(/\t/, $left_entry); + my ($right_acc, $right_seq, $right_qual) = split(/\t/, $right_entry); + + my $left_core = $left_acc; + $left_core =~ s|/1$||; + + my $right_core = $right_acc; + $right_core =~ s|/2$||; + + if ($left_core eq $right_core) { + ## write entries + + if (length($left_qual) == length($left_seq) + && + length($right_qual) == length($right_seq)) { + + ## AFAICT record looks good + + print $left_ofh join("\n", "\@$left_acc", $left_seq, "+", $left_qual) . "\n"; + print $right_ofh join("\n", "\@$right_acc", $right_seq, "+", $right_qual) . "\n"; + } + else { + print $left_broken_ofh $left_entry . "\n"; + print $right_broken_ofh $right_entry . "\n"; + } + + # reprime + $left_entry = <$left_fh>; + $right_entry = <$right_fh>; + } + else { + if ($left_core lt $right_core) { + print $left_broken_ofh $left_entry . "\n"; + $left_entry = <$left_fh>; + } + else { + print $right_broken_ofh $right_entry . "\n"; + $right_entry = <$right_fh>; + } + } +} + +close $left_ofh; +close $right_ofh; +close $left_broken_ofh; +close $right_broken_ofh; + + +exit(0); + + diff --git a/99.scripts/trinity_utils/util/misc/fastq_stats.pl b/99.scripts/trinity_utils/util/misc/fastq_stats.pl new file mode 100644 index 0000000..cd51489 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fastq_stats.pl @@ -0,0 +1,262 @@ +#!/usr/bin/env perl + +# written by +# +# Bob Freeman, Ph.D. +# Acorn Worm Informatics, Kirschner lab +# Dept of Systems Biology, Alpert 524 +# Harvard Medical School +# 200 Longwood Avenue +# Boston, MA 02115 +# 617/432.2294, vox +# +# bob_freeman@hms.harvard.edu + +=from_Bob + +I wrote a small script in Perl that gives a number of basic stats and can even generate a histogram of lengths for you. Script is attached. I typically use this for assessing the size / quality of my data. + +For example, after trimming a set of reads (prior to assembly), I used the command: + + fastq_stats.pl -i trimmed_reads.fastq -f "30,40,50,60,70,80,90,100" + +to assess resulting data, and my output is: + + Count27717551 + Sum2347820942 + Mean84.7052087 + + Min36 + Max101 + Median101 + + Q036 + Q154 + Q2101 + Q3101 + Q4101 + + 36(min) + <= 30.0:0 + <= 40.0:183537 + <= 50.0:283964 + <= 60.0:8209709 + <= 70.0:339854 + <= 80.0:458850 + <= 90.0:460218 + <= 100.0:1271776 + +I believe I've modified the file to identify fasta, fastq, and protein fasta ( *.fasta, *.fa, *.mfasta, *.pro, and *.pep). Output can be in table-format also ( -t ). And use the -h switch to print out the (scarce) help info. It does use Bioperl and the Perl modules Statistics::Descriptive. + +Hope you find it useful! + +Bob + +=cut + + +use strict; +use Bio::SeqIO; +use Carp; +use Data::Dumper; +use English qw ( -no_match_vars ); +use Statistics::Descriptive; + + +my $usage = qq ( + fastq_stats.pl -i infile -f partition|bins -t + + reads in fastq or fasta data and spits out some descripting stats + -i input file + -f frequency distribution, either # partitions or a quoted list of + comma-separated bins + -t print results in table (tab-delimited) format + +); + +# opens fastq or fasta data +# reads in each sequence, grabbing data about each +# then spits out output data + +$OUTPUT_AUTOFLUSH = 1; +our @seqlen_data; +our @hist_bins; +our ($infile, $freqdist, $fdref); +our $table = 0; + +parse_args(); +read_seq_data2(); +if (!$table) { + output_stats(); +} else { + output_stats_table(); +} + +exit; + +# +# +# END OF PROGRAM +# + +sub parse_args { + # + while (scalar(@ARGV)) { + my $arg = shift(@ARGV); + if ($arg eq '-h') { die $usage; } + elsif ($arg eq '-i') { $infile = shift(@ARGV); } + elsif ($arg eq '-f') { $freqdist = shift(@ARGV); } + elsif ($arg eq '-t') { $table = 1; } + else { die "unknown argument '$arg'"; } + } + + # check to ensure we have the required parameters + if (!defined $infile) { + print "Error from undefined input arguments!\n"; + die $usage; + } + + # parse frequency distribution stuff + if (defined $freqdist) { + $freqdist =~ s/ //g; + @hist_bins = split (/,/, $freqdist); + $freqdist = 1; + } else { + $freqdist = 0; + } +} + +sub read_seq_data { + + my ($file_suffix) = $infile =~ m/.+\.(\S+)$/; + # create input SeqIO object + my $seq_in = Bio::SeqIO->new( '-file' => "<$infile", + '-format' => $file_suffix ); + + while (my $inseq = $seq_in->next_seq()) { + push @seqlen_data, $inseq->length(); + + } #end of while +} + +sub read_seq_data2 { + + if ($infile =~ m/(\.fasta$)|(\.fa$)|(\.mfasta$)|(\.mfa$)|(\.pro$)|(\.pep$)/) { + read_fasta(); + } elsif ($infile =~ m/(\.fastq$)|(\.fq$)/) { + read_fastq(); + } else { + die "Error: Unknown file format for input file $infile!\n"; + } +} + +sub read_fasta { + # read in fasta data using BioPerl methods. These are pretty quick. + # create input SeqIO object + my $seq_in = Bio::SeqIO->new( '-file' => "<$infile", + '-format' => 'fasta' ); + + while (my $inseq = $seq_in->next_seq()) { + push @seqlen_data, $inseq->length(); + + } #end of while +} + +sub read_fastq { + # read in fastq data in old-fashioned method, as it's faster than using BioPerl + open (my $INFILE, "<$infile") || die "Error opening file $infile for reading: $!\n"; + while ( my $line = <$INFILE> ) { + #chomp $line; + #next if $line =~ m/^\s|#/; + if ($line =~ m/^@/) { + # skip to next line and push length + $line = <$INFILE>; + chomp $line; + push @seqlen_data, length($line); + } + } + close $INFILE; +} + +sub output_stats { + + print "\nSequence stats:\n"; + my $stat = Statistics::Descriptive::Full->new(); + $stat->add_data(@seqlen_data); + print "Count\t", $stat->count(), "\n"; + print "Sum\t", $stat->sum(), "\n"; + print "Mean\t", $stat->mean(), "\n\n"; + + print "Min\t", $stat->min(), "\n"; + print "Max\t", $stat->max(), "\n"; + print "Median\t", $stat->median(), "\n\n"; + print "Q0\t", $stat->quantile(0), "\n"; + print "Q1\t", $stat->quantile(1), "\n"; + print "Q2\t", $stat->quantile(2), "\n"; + print "Q3\t", $stat->quantile(3), "\n"; + print "Q4\t", $stat->quantile(4), "\n"; + + # do frequency distribution output if indicated + if ($freqdist) { + # call proper method + if (scalar(@hist_bins) == 1) { + $fdref = $stat->frequency_distribution_ref($hist_bins[0]); + } else { + $fdref = $stat->frequency_distribution_ref(\@hist_bins); + } + # now print it out + print "\n", $stat->min(), "(min)\n"; + for (sort {$a <=> $b} keys %$fdref) { + #sprintf "<= %.1f: %8u \n", $_, "<= $_, \tcount = $fdref->{$_}\n"; + printf "<= %5.1f:\t%8u \n", $_, $fdref->{$_}; + } + } +} + +sub output_stats_table { + # + # do calculations + my $stat = Statistics::Descriptive::Full->new(); + $stat->add_data(@seqlen_data); + if ($freqdist) { + # call proper method + if (scalar(@hist_bins) == 1) { + $fdref = $stat->frequency_distribution_ref($hist_bins[0]); + } else { + $fdref = $stat->frequency_distribution_ref(\@hist_bins); + } + } + + # print header + print "#Sequence stats:\n"; + print join("\t", "#","Count","Sum","Mean","Min","Max","Median","Q0","Q1","Q2","Q3","Q4"); + if ($freqdist) { + for (sort {$a <=> $b} keys %$fdref) { + #sprintf "<= %.1f: %8u \n", $_, "<= $_, \tcount = $fdref->{$_}\n"; + print "\t<=", $_; + } + } + print "\n"; + + #print results + print $infile, "\t"; + print $stat->count(), "\t"; + print $stat->sum(), "\t"; + print $stat->mean(), "\t"; + print $stat->min(), "\t"; + print $stat->max(), "\t"; + print $stat->median(), "\t"; + print $stat->quantile(0), "\t"; + print $stat->quantile(1), "\t"; + print $stat->quantile(2), "\t"; + print $stat->quantile(3), "\t"; + print $stat->quantile(4); + if ($freqdist) { + # now print it out + for (sort {$a <=> $b} keys %$fdref) { + #sprintf "<= %.1f: %8u \n", $_, "<= $_, \tcount = $fdref->{$_}\n"; + print "\t", $fdref->{$_}; + } + } + print "\n"; +} diff --git a/99.scripts/trinity_utils/util/misc/fastq_unweave_pairs.pl b/99.scripts/trinity_utils/util/misc/fastq_unweave_pairs.pl new file mode 100644 index 0000000..b69a93d --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/fastq_unweave_pairs.pl @@ -0,0 +1,52 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use Fastq_reader; + +my $usage = "\n\nusage: $0 interleaved.fq [left_output_filename.fq right_output_filename.fq]\n\n"; + +my $interleaved_fq = $ARGV[0] or die $usage; +my $left_out_filename = $ARGV[1] || "unweaved.left.$$.fq"; +my $right_out_filename = $ARGV[2] || "unweaved.right.$$.fq"; + +main: { + + my $fastq_reader = new Fastq_reader($interleaved_fq); + + open (my $left_ofh, ">$left_out_filename") or die "Error, cannot write to $left_out_filename"; + open (my $right_ofh, ">$right_out_filename") or die "Error, cannot write to $right_out_filename"; + + while (my $fq = $fastq_reader->next()) { + + my $read_name = $fq->get_full_read_name(); + + my $record = $fq->get_fastq_record(); + + if ($read_name =~ m|/1$|) { + print $left_ofh $record; + } + elsif ($read_name =~ m|/2$|) { + print $right_ofh $record; + } + else { + die "Error, cannot decipher left or right read from name: $read_name"; + } + } + + close $left_ofh; + close $right_ofh; + + + print "Done.\n"; + + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/filter_out_accs_from_fasta.pl b/99.scripts/trinity_utils/util/misc/filter_out_accs_from_fasta.pl new file mode 100644 index 0000000..9da2746 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/filter_out_accs_from_fasta.pl @@ -0,0 +1,64 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use Data::Dumper; + + +unless ($ARGV[0] && $ARGV[1]) { + die "Usage: $0 \$fastaFile \$acc_listing\n"; +} + + +my $fastaFile = $ARGV[0]; +my $acc_listing = $ARGV[1]; + +open (ACC, "$acc_listing"); +my %accs; +while () { + chomp; + my @x = split (/\s+/); + foreach my $acc (@x) { + if ($acc =~ /\w/) { + $acc =~ s/\s//g; + $acc =~ s/\W/_/g; + $accs{$acc} = 1; + } + } +} +close ACC; + + + +open (FASTA, "$fastaFile"); + +my %to_delete = %accs; + +my $ok_flag = 1; +while () { + if (/^>(\S+)/) { + my $acc = $1; + if ($accs{$acc}) { + # targeted for skipping + $ok_flag = 0; + delete $to_delete{$acc}; + print STDERR "-found and skipping $acc\n"; + } + else { + # not targeted for skipping + $ok_flag = 1; + } + } + if ($ok_flag) { + print; + } +} + +if (%to_delete) { + confess "Error, didn't observe and exclude the following accessions from the fasta file: " . Dumper(\%to_delete); +} + +print STDERR "-done.\n\n"; + +exit(0); diff --git a/99.scripts/trinity_utils/util/misc/filter_similar_seqs_expr_and_strand_aware.pl b/99.scripts/trinity_utils/util/misc/filter_similar_seqs_expr_and_strand_aware.pl new file mode 100644 index 0000000..982a04b --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/filter_similar_seqs_expr_and_strand_aware.pl @@ -0,0 +1,304 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use Getopt::Long qw(:config posix_default no_ignore_case bundling pass_through); +use Data::Dumper; +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use Fasta_reader; + +my $help_flag; + + +my $MIN_PCT_MAX_EXPR = 5; + +my $CPU = 2; + +my $REF_MIN_PCT_LEN = 90; + +my $usage = <<__EOUSAGE__; + + +Algorithm is as follows: + + Run CDHIT to cluster according to sequence identity and length + + A reference sequence is selected as having the highest expression level and within ${REF_MIN_PCT_LEN} % length of the longest sequence in that cluster. + + Any sequences having less than $MIN_PCT_MAX_EXPR of the selected reference sequence are filtered out. + + Use the --filter_antisense_only flag to only have filtering applied to sequences with opposite transcribed orientation from the selected reference sequence in each cluster. + + Output is a new fasta file containing only those entries that pass the filtering criteria. + + +################################################################ +# +# --transcripts_fasta target transcript fasta file +# +# --expr_matrix transcript expression matrix (TPM matrix) +# +# +# Optional: +# +# --min_pct_max_expr default: $MIN_PCT_MAX_EXPR +# +# --filter_antisense_only default: off +# +# --CPU number of threads +# +# --ref_min_pct_len minimum percent of max length to define candidate reference sequences +# default: $REF_MIN_PCT_LEN +# +########################################################################### + + +__EOUSAGE__ + + ; + + +# cd-hit-est -i axo.indropB.iworm.fa -o axo.indropB.iworm.fa.cdhit98.fa -c 0.98 -aS 0.95 -T 20 + +my $transcripts_fasta_file; +my $expr_matrix_file; +my $filter_antisense_only; + +my $DEBUG = 0; + + +&GetOptions ( 'h' => \$help_flag, + 'transcripts_fasta=s' => \$transcripts_fasta_file, + 'expr_matrix=s' => \$expr_matrix_file, + 'filter_antisense_only' => \$filter_antisense_only, + + 'CPU=i' => \$CPU, + 'min_pct_max_expr=i' => \$MIN_PCT_MAX_EXPR, + 'REF_MIN_PCT_LEN=i' => \$REF_MIN_PCT_LEN, + + + 'd' => \$DEBUG, + ); + + +if ($help_flag) { + die $usage; +} + + +unless($transcripts_fasta_file && $expr_matrix_file) { + die $usage; +} + + +main: { + + my $cdhit_prefix = "$transcripts_fasta_file.cdhit"; + + my $cmd = "cd-hit-est -i $transcripts_fasta_file -o $cdhit_prefix -d 0 -c 0.98 -aS 0.95 -T $CPU"; + unless (-e "$cdhit_prefix.ok") { + &process_cmd($cmd); + + &process_cmd("touch $cdhit_prefix.ok"); + } + + my %top_expr_val = &get_top_expr_val($expr_matrix_file); + + my %accs_retain = &filter_noisy_transcripts("$cdhit_prefix.clstr", \%top_expr_val, $REF_MIN_PCT_LEN, + $MIN_PCT_MAX_EXPR, $filter_antisense_only); + + # output refined fasta file: + + + my $fasta_reader = new Fasta_reader($transcripts_fasta_file); + + my $count_kept = 0; + my $count_excluded = 0; + + while (my $seq_obj = $fasta_reader->next()) { + my $acc = $seq_obj->get_accession(); + + if ($accs_retain{$acc}) { + print $seq_obj->get_FASTA_format(); + delete $accs_retain{$acc}; + $count_kept++; + } + else { + $count_excluded++; + } + } + + if (%accs_retain) { + die "Error, didnt retrieve sequences for accession: " . Dumper(%accs_retain); + } + else { + print STDERR "Done. Excluded $count_excluded = " . sprintf("%.2f", $count_excluded / ($count_kept + $count_excluded) * 100) . "% of sequences\n"; + } + + exit(0); +} + +#### +sub get_top_expr_val { + my ($expr_matrix_file) = @_; + + my %top_expr_val; + + open(my $fh, $expr_matrix_file) or die "Error, cannot open file: $expr_matrix_file"; + my $header = <$fh>; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my $trans_acc = shift @x; + @x = sort {$a<=>$b} @x; + my $max_expr = pop @x; + $top_expr_val{$trans_acc} = $max_expr; + + } + + close $fh; + + return(%top_expr_val); +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret"; + } + + return; +} + +#### +sub filter_noisy_transcripts { + my ($cdhit_clstr_file, $top_expr_vals_href, $ref_min_pct_len, $min_pct_max_expr, $filter_antisense_only) = @_; + + + + + my %accs_want; + + my $select_clusters_want_sref = sub { + + my @structs = @_; + + + + @structs = reverse sort {$a->{len}<=>$b->{len}} @structs; + + my $longest_len = $structs[0]->{len}; + + my @ref_structs; + foreach my $struct (@structs) { + if ($struct->{len} / $longest_len * 100 >= $ref_min_pct_len) { + push (@ref_structs, $struct); + } + } + + # take the highest expr entry as official ref seq + @ref_structs = reverse sort {$a->{expr}<=>$b->{expr}} @ref_structs; + + + my $ref_seq_struct = shift @ref_structs; + $accs_want{ $ref_seq_struct->{acc} } = 1; + + my $audit_text = "* ref selected as: $ref_seq_struct->{acc} " + . " len: $ref_seq_struct->{len}" + . " expr: $ref_seq_struct->{expr}" + . " [$ref_seq_struct->{orient}]\n"; + + + print STDERR "* selected ref seq as: " . Dumper($ref_seq_struct) if $DEBUG; + + + my $min_allowed_expr = $ref_seq_struct->{expr} * $min_pct_max_expr / 100; + + my $excluded_flag = 0; + + foreach my $struct (@structs) { + + if ($struct eq $ref_seq_struct) { next; } + + my $retain_flag = 0; + + if ($filter_antisense_only && $struct->{orient} eq $ref_seq_struct->{orient}) { + # all good, not antisense + $retain_flag = 1; + } + + elsif ($struct->{expr} >= $min_allowed_expr) { + # meets expression criteria + $retain_flag = 1; + } + + if ($retain_flag) { + $accs_want{ $struct->{acc} } = 1; + $audit_text .= "-keeping: $struct->{acc} len: $struct->{len} expr: $struct->{expr} [$struct->{orient}]\n"; + } + else { + print STDERR "-excluding " . Dumper($struct) if $DEBUG; + $audit_text .= "-EXCLUDING: $struct->{acc} len: $struct->{len} expr: $struct->{expr} [$struct->{orient}]\n"; + + $excluded_flag = 1; + } + } + + if ($excluded_flag) { + print STDERR "$audit_text\n"; + } + + }; + + + my @cluster; + open(my $fh, $cdhit_clstr_file) or die $!; + while (<$fh>) { + chomp; + if (/^>/) { + if (@cluster) { + &$select_clusters_want_sref(@cluster); + } + @cluster = (); + } + else { + my ($idx, $len, $trans_acc, $selected_star, $orient_pct) = split(/\s+/); + $len =~ s/nt,//; + $trans_acc =~ s/>//; + $trans_acc =~ s/\.\.\.$//; + + my $expr = $top_expr_vals_href->{$trans_acc}; + unless (defined $expr) { + die "Error, no expr value for $trans_acc"; + } + + my ($orient, $pct) = ('+', '.'); + if ($selected_star eq 'at') { + ($orient, $pct) = split(/\//, $orient_pct); + } + + my $struct = { acc => $trans_acc, + len => $len, + orient => $orient, + expr => $expr, + }; + push (@cluster, $struct); + + } + + } + close $fh; + + if (@cluster) { + &$select_clusters_want_sref(@cluster); + } + + return(%accs_want); +} diff --git a/99.scripts/trinity_utils/util/misc/flattened_gff_n_genome_to_Trinity_emulator.pl b/99.scripts/trinity_utils/util/misc/flattened_gff_n_genome_to_Trinity_emulator.pl new file mode 100644 index 0000000..fa3eb42 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/flattened_gff_n_genome_to_Trinity_emulator.pl @@ -0,0 +1,171 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use Fasta_reader; +use Nuc_translator; +use Data::Dumper; + +use Getopt::Long qw(:config posix_default no_ignore_case bundling pass_through); + + +my $usage = <<__EOUSAGE__; + +################################################### +# +# Required: +# +# --genome_fa genome fasta file +# +# --flattened_gff flattened gff file +# +# Optional: +# +# --no_revcomp do not revcomp reverse strand transcripts +# +################################################### + + +__EOUSAGE__ + + ; + + +my $genome_fa; +my $flattened_gff; +my $NO_REVCOMP_FLAG = 0; +my $help_flag; + +&GetOptions ( 'h' => \$help_flag, + + 'genome_fa=s' => \$genome_fa, + 'flattened_gff=s' => \$flattened_gff, + + 'no_revcomp' => \$NO_REVCOMP_FLAG, + ); + + +if ($help_flag) { + die $usage; +} + +unless ($genome_fa && $flattened_gff) { + die $usage; +} + +main: { + + + my $fasta_reader = new Fasta_reader($genome_fa); + my %seqs = $fasta_reader->retrieve_all_seqs_hash(); + + my %gene_to_transcripts = &parse_gff($flattened_gff); + + unless (%gene_to_transcripts) { + die "Error, no gff records parsed. Be sure to use the flattened gff file"; + } + + #die Dumper(\%gene_to_transcripts); + + foreach my $gene (keys %gene_to_transcripts) { + + my @transcripts = keys %{$gene_to_transcripts{$gene}}; + my $transcript_counter = 0; + foreach my $transcript (@transcripts) { + $transcript_counter += 1; + my @regions = @{$gene_to_transcripts{$gene}->{$transcript}}; + @regions = sort {$a->{lend}<=>$b->{lend}} @regions; + my $trans_seq = ""; + my @path_coords; + foreach my $region (@regions) { + my $lend = $region->{lend}; + my $rend = $region->{rend}; + my $chr = $region->{chr}; + my $exon_number = $region->{exon_number}; + my $seg_len = $rend - $lend + 1; + my $seq_seg = substr($seqs{$chr}, $lend-1, $seg_len); + + push (@path_coords, [$exon_number, length($trans_seq)+1, length($trans_seq) + length($seq_seg)]); + $trans_seq .= $seq_seg; + } + if ($regions[0]->{orient} eq '-' && ! $NO_REVCOMP_FLAG) { + # revcomp everthing + $trans_seq = &reverse_complement($trans_seq); + my $trans_len = length($trans_seq); + foreach my $path_coord (@path_coords) { + my $new_rend = $trans_len - $path_coord->[1] + 1; + my $new_lend = $trans_len - $path_coord->[2] + 1; + + ($path_coord->[1], $path_coord->[2]) = ($new_lend, $new_rend); + } + @path_coords = sort {$a->[1]<=>$b->[1]} @path_coords; + } + + ## construct header + my @path_text; + foreach my $path_coord (@path_coords) { + my ($exon_number, $lend, $rend) = @$path_coord; + $lend--; + $rend--; + push (@path_text, "$exon_number:$lend-$rend"); + } + + + my $header = ">${gene}_i${transcript_counter} path=[" . join(" ", @path_text) . "]"; + + print "$header\n$trans_seq\n"; + } + } + + exit(0); + + +} + + +#### +sub parse_gff { + my ($gff_file) = @_; + + my %gene_to_transcripts; + + open(my $fh, $gff_file) or die $!; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my $chr = $x[0]; + my $feat_type = $x[2]; + my $lend = $x[3]; + my $rend = $x[4]; + my $orient = $x[6]; + my $info = $x[8]; + + unless ($feat_type eq "exonic_part") { next; } + + $info =~ /transcripts \"([^\"]+)\";.*\s+exonic_part_number \"(\d+)\"; gene_id \"([^\"]+)\"/ or die "Error, cannot parse $info"; + + my $transcripts_list = $1; + my $exonic_part_number = $2; + my $gene_id = $3; + + my $region_struct = { chr => $chr, + lend => $lend, + rend => $rend, + orient => $orient, + exon_number => $exonic_part_number, + }; + + my @transcripts = split(/\+/, $transcripts_list); + foreach my $transcript (@transcripts) { + push (@{$gene_to_transcripts{$gene_id}->{$transcript}}, $region_struct); + } + } + + close $fh; + + return(%gene_to_transcripts); +} + + diff --git a/99.scripts/trinity_utils/util/misc/frag_boundary_to_wig.pl b/99.scripts/trinity_utils/util/misc/frag_boundary_to_wig.pl new file mode 100644 index 0000000..dbb7cb0 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/frag_boundary_to_wig.pl @@ -0,0 +1,62 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 file.frag_coords (LEND|REND)\n\n"; + +my $file = $ARGV[0] or die $usage; +my $end = $ARGV[1] or die $usage; + +unless ($end eq "LEND" || $end eq "REND") { + die $usage; +} + +main: { + + my @histogram; + my $curr_scaffold = ""; + + open (my $fh, $file) or die $usage; + while (<$fh>) { + chomp; + my ($scaff, $read, $lend, $rend) = split(/\t/); + if ($curr_scaffold ne $scaff) { + &write_wig($curr_scaffold, \@histogram) if @histogram; + + # re-init + $curr_scaffold = $scaff; + @histogram = (); + } + + my $pos = ($end eq "LEND") ? $lend : $rend; + + + $histogram[$pos]++; + } + close $fh; + + if (@histogram) { + &write_wig($curr_scaffold, \@histogram); + } + + exit(0); +} + +#### +sub write_wig { + my ($scaffold, $histogram_aref) = @_; + + print "variableStep chrom=$scaffold\n"; + + for (my $i = 1; $i <= $#$histogram_aref; $i++) { + my $val = $histogram_aref->[$i]; + if (defined $val) { + print join("\t", $i, $val) . "\n"; + } + } + + return; +} + + diff --git a/99.scripts/trinity_utils/util/misc/frag_to_bed.pl b/99.scripts/trinity_utils/util/misc/frag_to_bed.pl new file mode 100644 index 0000000..6ff5d10 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/frag_to_bed.pl @@ -0,0 +1,32 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; + +use lib ("$ENV{EUK_MODULES}"); +use Gene_obj; + + +while (<>) { + chomp; + my ($scaff, $read_acc, $lend, $rend) = split(/\t/); + + my $gene_obj = new Gene_obj(); + $gene_obj->populate_gene_object({}, {$lend => $rend}); + $gene_obj->{asmbl_id} = $scaff; + + $gene_obj->{TU_feat_name} = $read_acc; + $gene_obj->{Model_feat_name} = $read_acc; + + + print $gene_obj->to_BED_format(); + + + +} + +exit(0); + + + diff --git a/99.scripts/trinity_utils/util/misc/gene_gff3_to_introns.pl b/99.scripts/trinity_utils/util/misc/gene_gff3_to_introns.pl new file mode 100644 index 0000000..cd96479 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/gene_gff3_to_introns.pl @@ -0,0 +1,69 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Gene_obj; +use Fasta_reader; +use GFF3_utils; +use Carp; + + +my $usage = "usage: $0 genes.gff3 genome.fasta\n\n"; + +my $genes_gff3 = $ARGV[0] or die $usage; +my $genome_fasta_file = $ARGV[1] or die $usage; + + +my $fasta_reader = new Fasta_reader($genome_fasta_file); +my %genome = $fasta_reader->retrieve_all_seqs_hash(); + +my $gene_obj_indexer_href = {}; + +## associate gene identifiers with contig id's. +my $contig_to_gene_list_href = &GFF3_utils::index_GFF3_gene_objs($genes_gff3, $gene_obj_indexer_href); + +foreach my $asmbl_id (sort keys %$contig_to_gene_list_href) { + + my $genome_seq = $genome{$asmbl_id} or die "Error, cannot find sequence for $asmbl_id"; #cdbyank_linear($asmbl_id, $fasta_db); + + my @gene_ids = @{$contig_to_gene_list_href->{$asmbl_id}}; + + foreach my $gene_id (@gene_ids) { + my $gene_obj_ref = $gene_obj_indexer_href->{$gene_id}; + + my $orientation = $gene_obj_ref->get_orientation(); + + my $intron_text = ""; + + foreach my $isoform ($gene_obj_ref, $gene_obj_ref->get_additional_isoforms()) { + + my @intron_coords = $isoform->get_intron_coordinates(); + + + my $trans_id = $isoform->{Model_feat_name}; + + foreach my $intron (@intron_coords) { + my ($end5, $end3) = @$intron; + my ($lend, $rend) = sort {$a<=>$b} ($end5, $end3); + my $left_splice_dinuc = substr($genome_seq, $lend-1, 2); + my $right_splice_dinuc = substr($genome_seq, $rend-1-1, 2); + + $intron_text .= join("\t", $gene_id, $trans_id, $asmbl_id, "$lend-$rend", $orientation, "$left_splice_dinuc..$right_splice_dinuc") . "\n"; + } + } + + if ($intron_text) { + print "$intron_text\n"; + } + + + + + } +} + + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/gene_to_shared_transcript_content.pl b/99.scripts/trinity_utils/util/misc/gene_to_shared_transcript_content.pl new file mode 100644 index 0000000..2ba268a --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/gene_to_shared_transcript_content.pl @@ -0,0 +1,81 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use lib ($ENV{EUK_MODULES}); +use Fasta_reader; +use Nuc_translator; + +my $usage = "usage:\t $0 refTrans.fasta [strandSpecific=0]\n\n"; + +my $trans_fa = $ARGV[0] or die $usage; + +my $strand_specific_flag = $ARGV[1] or 0; + +my $kmer_size = 24; + +main: { + + + my %kmer_to_gene; + + my $counter = 0; + my $fasta_reader = new Fasta_reader($trans_fa); + while (my $seq_obj = $fasta_reader->next()) { + + my $accession = $seq_obj->get_accession(); + my $sequence = uc $seq_obj->get_sequence(); + + $counter++; + print STDERR "\r[$counter] -tracking $accession"; + + + my ($trans, $gene) = split(/;/, $accession); + unless ($gene) { + die "Error, need format: >trans;gene , instead found: $accession "; + } + + for (my $i = 0; $i <= length($sequence) - $kmer_size; $i++) { + + my $kmer = substr($sequence, $i, $kmer_size); + + unless (length($kmer) == $kmer_size) { + die "Error, didn't extract proper kmer length $kmer_size : [$kmer] "; + } + + unless ($strand_specific_flag) { + my @kmers = ($kmer, &reverse_complement($kmer)); + @kmers = sort @kmers; + $kmer = $kmers[0]; + } + + $kmer_to_gene{$kmer}->{$gene} = 1; + } + } + + print STDERR "\n\n-reorganizing as gene sets to kmer lists.\n"; + + my %genes_to_kmers; + foreach my $kmer (keys %kmer_to_gene) { + + my @genes = keys %{$kmer_to_gene{$kmer}}; + + if (scalar @genes > 1) { + my $gene_token = join(",", sort @genes); + push (@{$genes_to_kmers{$gene_token}}, $kmer); + } + } + + print STDERR "\n\n-outputting gene links and kmers.\n"; + + foreach my $gene_set (sort keys %genes_to_kmers) { + + my @kmers = sort @{$genes_to_kmers{$gene_set}}; + + print "$gene_set\t" . join(",", sort @kmers) . "\n"; + } + + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/misc/genome_gff3_to_gene_gff3_partitions.pl b/99.scripts/trinity_utils/util/misc/genome_gff3_to_gene_gff3_partitions.pl new file mode 100644 index 0000000..4297916 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/genome_gff3_to_gene_gff3_partitions.pl @@ -0,0 +1,113 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$ENV{TRINITY_HOME}/PerlLib"); +use Gene_obj; +use Fasta_reader; +use GFF3_utils; +use Carp; +use Nuc_translator; + + +my $MIN_LENGTH = 1000; +my $MIN_ISO_COUNT = 2; + +my $usage = "\n\nusage: $0 gff3_file genome_db [flank]\n\n"; + +my $gff3_file = $ARGV[0] or die $usage; +my $fasta_db = $ARGV[1] or die $usage; +my $flank = $ARGV[2] || 0; + +my $fasta_reader = new Fasta_reader($fasta_db); +my %genome = $fasta_reader->retrieve_all_seqs_hash(); + +my $gene_obj_indexer_href = {}; + +## associate gene identifiers with contig id's. +my $contig_to_gene_list_href = &GFF3_utils::index_GFF3_gene_objs($gff3_file, $gene_obj_indexer_href); + +my $outdir = "gene_contigs"; +mkdir $outdir or die "Error, $outdir already exists"; +my $gene_contigs_per_bin = 100; +my $gene_counter = 0; + + +foreach my $asmbl_id (sort keys %$contig_to_gene_list_href) { + + my $genome_seq = $genome{$asmbl_id} or die "Error, cannot find sequence for $asmbl_id"; #cdbyank_linear($asmbl_id, $fasta_db); + + my @gene_ids = @{$contig_to_gene_list_href->{$asmbl_id}}; + + foreach my $gene_id (@gene_ids) { + + my $gene_id_for_filename = $gene_id; + $gene_id_for_filename =~ s/\W/_/g; + + my $gene_obj_ref = $gene_obj_indexer_href->{$gene_id}; + + my ($gene_contig_lend, $gene_contig_rend) = sort {$a<=>$b} $gene_obj_ref->get_gene_span(); + + $gene_contig_lend -= $flank; + $gene_contig_rend += $flank; + + + $gene_obj_ref->adjust_gene_coordinates(-1 * ($gene_contig_lend-1)); + + # update gene identifiers. + my @iso_objs; + foreach my $iso_obj ($gene_obj_ref, $gene_obj_ref->get_additional_isoforms()) { + $iso_obj->{asmbl_id} = $gene_id; + my $cdna_len = $iso_obj->get_cDNA_length(); + if ($cdna_len >= $MIN_LENGTH) { + push (@iso_objs, $iso_obj); + } + } + + if (scalar @iso_objs >= $MIN_ISO_COUNT) { + + $gene_obj_ref->delete_isoforms(); + + my $gene_obj = shift @iso_objs; + $gene_obj->add_isoform(@iso_objs); + + + + + $gene_counter++; + my $bindir = "$outdir/bin_" . int($gene_counter/$gene_contigs_per_bin) . "/$gene_id_for_filename"; + print STDERR "[$gene_counter] -processing $bindir\n"; + &process_cmd("mkdir -p $bindir"); + + my $contig_filename = "$bindir/gene.fa"; + open (my $ofh, ">$contig_filename") or die $!; + my $gene_seq = substr($genome_seq, $gene_contig_lend-1, $gene_contig_rend - $gene_contig_lend + 1); + print $ofh ">$gene_id\n$gene_seq\n"; + close $ofh; + + my $gff3_filename = "$bindir/gene.gff3"; + open ($ofh, ">$gff3_filename") or die $!; + print $ofh $gene_obj->to_GFF3_format(); + close $ofh; + } + + if ($gene_counter > 10) { last; } + + } +} + + +exit(0); + +#### +sub process_cmd { + my ($cmd) = @_; + + my $ret = system($cmd); + if ($ret) { + die "Error, CMD: $cmd died with ret $ret"; + } + + return; +} diff --git a/99.scripts/trinity_utils/util/misc/get_GC_content_dist.pl b/99.scripts/trinity_utils/util/misc/get_GC_content_dist.pl new file mode 100644 index 0000000..218b5c1 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/get_GC_content_dist.pl @@ -0,0 +1,65 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use Fasta_reader; +use Process_cmd; + + +my $usage = "\n\n\tusage: $0 Trinity.fasta [out_prefix='GC_content']\n\n"; + +my $fasta_file = $ARGV[0] or die $usage; +my $out_prefix = $ARGV[1] || "GC_content"; + +main: { + + my $dat_out_file = "$out_prefix.dat"; + + open(my $ofh, ">$dat_out_file") or die "Error, cannot write to $dat_out_file"; + + my $fasta_reader = new Fasta_reader($fasta_file); + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + my $sequence = $seq_obj->get_sequence(); + + my $seq_len = length($sequence); + + my $gc_count = 0; + while ($sequence =~ /[gc]/ig) { + $gc_count++; + } + my $pct_gc = sprintf("%.2f", $gc_count / $seq_len * 100); + + print $ofh join("\t" ,$acc, $pct_gc) . "\n"; + } + + # generate histogram + my $R_code = <<__eoR__; + + data = read.table("$dat_out_file", header=F, row.names=1) + pdf("$dat_out_file.hist.pdf") + hist(data[,1], br=100) + message("\n\nmean: ", sprintf("%.2f", mean(data[,1])), ", median: ", median(data[,1]), "\n\n") + dev.off() + +__eoR__ + +; + + { + my $Rscript_file = "$out_prefix.R"; + open (my $ofh, ">$Rscript_file") or die "Error, cannot write to file: $Rscript_file"; + print $ofh $R_code; + close $ofh; + + # run it + my $cmd = "Rscript $Rscript_file"; + &process_cmd($cmd); + } + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/misc/get_longest_isoform_seq_per_trinity_gene.pl b/99.scripts/trinity_utils/util/misc/get_longest_isoform_seq_per_trinity_gene.pl new file mode 100644 index 0000000..53b35bb --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/get_longest_isoform_seq_per_trinity_gene.pl @@ -0,0 +1,62 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + +print STDERR "\n\n\tNOTE - longest transcript isn't always the best transcript!... consider filtering based on relative expression support ... \n\n"; + +my $usage = "usage: $0 Trinity.fasta\n\n"; + +my $trin_fasta = $ARGV[0] or die $usage; + +main: { + my %gene_to_longest_transcript; + + my $fasta_reader = new Fasta_reader($trin_fasta); + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + my $header = $seq_obj->get_header(); + + my $gene_id; + + if ($acc =~ /^(.*comp\d+_c\d+)_seq/) { + $gene_id = $1; + } + elsif ($acc =~ /^(.*c\d+_g\d+)_i/) { + $gene_id = $1; + } + + unless ($gene_id) { + die "Error, cannot parse gene identifier from acc: $acc"; + } + + my $sequence = $seq_obj->get_sequence(); + + if ( (! exists $gene_to_longest_transcript{$gene_id}) + || + $gene_to_longest_transcript{$gene_id}->{length} < length($sequence)) { + + $gene_to_longest_transcript{$gene_id} = { length => length($sequence), + acc => $acc, + sequence => $sequence, + header => $header, + }; + } + + + } + + foreach my $longest_seq ( reverse sort {$a->{length}<=>$b->{length}} values %gene_to_longest_transcript) { + + print ">" . $longest_seq->{header} . "\n" . $longest_seq->{sequence} . "\n"; + + } + + print STDERR "\n\nok. Done.\n\n"; + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/misc/get_path_nodes_from_fasta.pl b/99.scripts/trinity_utils/util/misc/get_path_nodes_from_fasta.pl new file mode 100644 index 0000000..bf36a3d --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/get_path_nodes_from_fasta.pl @@ -0,0 +1,44 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my @entries; + +while (<>) { + + if (/>(\w+).*path=\[(.*)\]/) { + + my $acc = $1; + my $path = $2; + + my @node_descr = split(/\s+/, $path); + + my @nodes; + + foreach my $node (@node_descr) { + $node =~ s/:.*$//; + push (@nodes, $node); + } + + push (@entries, { + acc => $acc, + nodes => join(" ", @nodes), + }); + } +} + + +@entries = sort {$a->{nodes} cmp $b->{nodes}} @entries; + + +foreach my $entry (@entries) { + print join("\t", $entry->{acc}, $entry->{nodes}) . "\n"; +} + + + +exit(0); + + + diff --git a/99.scripts/trinity_utils/util/misc/get_welds_from_chrysals_graphFromFasta_out.pl b/99.scripts/trinity_utils/util/misc/get_welds_from_chrysals_graphFromFasta_out.pl new file mode 100644 index 0000000..2cbff21 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/get_welds_from_chrysals_graphFromFasta_out.pl @@ -0,0 +1,20 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +while (<>) { + if (/Welding: >(a\d+;\d+).*to >(a\d+;\d+)/) { + my $from = $1; + my $to = $2; + print join("\t", sort($from, $to)) . "\n"; + } + elsif (/SCAFFOLD_ACCEPT:\s+(a\d+;\d+)\s+\d+\s+(a\d+;\d+)/) { + my $from = $1; + my $to = $2; + print join("\t", sort($from, $to)) . "\n"; + } +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/gff3_file_to_cdna.pl b/99.scripts/trinity_utils/util/misc/gff3_file_to_cdna.pl new file mode 100644 index 0000000..4b6a8d9 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/gff3_file_to_cdna.pl @@ -0,0 +1,107 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Gene_obj; +use Fasta_reader; +use GFF3_utils; +use Carp; +use Nuc_translator; + +my $usage = "\n\nusage: $0 gff3_file genome_fasta [flank=0]\n\n"; + +my $gff3_file = $ARGV[0] or die $usage; +my $fasta_db = $ARGV[1] or die $usage; +my $flank = $ARGV[2] || 0; + +my ($upstream_flank, $downstream_flank) = (0,0); + +if ($flank) { + if ($flank =~ /:/) { + ($upstream_flank, $downstream_flank) = split (/:/, $flank); + } + else { + ($upstream_flank, $downstream_flank) = ($flank, $flank); + } +} + +if ($upstream_flank < 0 || $downstream_flank < 0) { + die $usage; +} + + + +my $fasta_reader = new Fasta_reader($fasta_db); +my %genome = $fasta_reader->retrieve_all_seqs_hash(); + +my $gene_obj_indexer_href = {}; + +## associate gene identifiers with contig id's. +my $contig_to_gene_list_href = &GFF3_utils::index_GFF3_gene_objs($gff3_file, $gene_obj_indexer_href); + +foreach my $asmbl_id (sort keys %$contig_to_gene_list_href) { + + my $genome_seq = $genome{$asmbl_id} or die "Error, cannot find sequence for $asmbl_id"; #cdbyank_linear($asmbl_id, $fasta_db); + + my @gene_ids = @{$contig_to_gene_list_href->{$asmbl_id}}; + + foreach my $gene_id (@gene_ids) { + my $gene_obj_ref = $gene_obj_indexer_href->{$gene_id}; + + foreach my $isoform ($gene_obj_ref, $gene_obj_ref->get_additional_isoforms()) { + + my $orientation = $isoform->get_orientation(); + my ($model_lend, $model_rend) = sort {$a<=>$b} $isoform->get_model_span(); + my ($gene_lend, $gene_rend) = sort {$a<=>$b} $isoform->get_gene_span(); + + my $isoform_id = $isoform->{Model_feat_name}; + + my $seq = $isoform->create_cDNA_sequence(\$genome_seq); + if ($upstream_flank || $downstream_flank) { + $seq = &add_flank($seq, $upstream_flank, $downstream_flank, $gene_lend, $gene_rend, $orientation, \$genome_seq); + } + + unless ($seq) { + print STDERR "-warning, no cDNA sequence for $isoform_id\n"; + next; + } + + $seq =~ s/(\S{60})/$1\n/g; # make fasta format + chomp $seq; + + my $com_name = $isoform->{com_name} || ""; + + print ">$gene_id" . "::" . "$isoform_id\n$seq\n"; + } + } +} + + +exit(0); + + +#### +sub add_flank { + my ($seq, $upstream_flank, $downstream_flank, $lend, $rend, $orientation, $genome_seq_ref) = @_; + + my $far_left = ($orientation eq '+') ? $lend - $upstream_flank : $lend - $downstream_flank; + + if ($far_left < 1) { $far_left = 1; } + + my $flank_right = ($orientation eq '+') ? $downstream_flank : $upstream_flank; + + my $left_seq = substr($$genome_seq_ref, $far_left - 1, $lend - $far_left); + + my $right_seq = substr($$genome_seq_ref, $rend, $flank_right); + + if ($orientation eq '+') { + return (lc($left_seq) . uc($seq) . lc($right_seq)); + } + else { + return (lc(&reverse_complement($right_seq)) . uc($seq) . lc(&reverse_complement($left_seq))); + } +} + + diff --git a/99.scripts/trinity_utils/util/misc/gff3_file_utr_coverage_trimmer.pl b/99.scripts/trinity_utils/util/misc/gff3_file_utr_coverage_trimmer.pl new file mode 100644 index 0000000..632a3e0 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/gff3_file_utr_coverage_trimmer.pl @@ -0,0 +1,234 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Gene_obj; +use Fasta_reader; +use GFF3_utils; +use Carp; +use Nuc_translator; +use Data::Dumper; + +my $usage = "\n\nusage: $0 gff3_file utr_trim_dat [DEBUG]\n\n"; + +my $gff3_file = $ARGV[0] or die $usage; +my $utr_trim_dat = $ARGV[1] or die $usage; +my $DEBUG = $ARGV[2] || 0; + +my %utr_trim_info; +{ + open (my $fh, $utr_trim_dat) or die $!; + while (<$fh>) { + chomp; + my ($acc, $coords, @rest) = split(/\t/); + my ($lend, $rend) = split(/-/, $coords); + $utr_trim_info{$acc} = [$lend, $rend]; + } + close $fh; +} + + +my $gene_obj_indexer_href = {}; + +## associate gene identifiers with contig id's. +my $contig_to_gene_list_href = &GFF3_utils::index_GFF3_gene_objs($gff3_file, $gene_obj_indexer_href); + +foreach my $asmbl_id (sort keys %$contig_to_gene_list_href) { + + my @gene_ids = @{$contig_to_gene_list_href->{$asmbl_id}}; + + foreach my $gene_id (@gene_ids) { + my $gene_obj_ref = $gene_obj_indexer_href->{$gene_id}; + + my $trimmed_flag = 0; + + foreach my $isoform ($gene_obj_ref, $gene_obj_ref->get_additional_isoforms()) { + + if ($isoform->has_UTRs()) { + + my $isoform_acc = $isoform->{Model_feat_name}; + + if (my $trim_coords_aref = $utr_trim_info{$isoform_acc}) { + + my $cdna_len = $isoform->get_cDNA_length(); + + my $targeted_cdna_len = $trim_coords_aref->[1] - $trim_coords_aref->[0] + 1; + + + print "\n====\nBefore:\n" . $isoform->to_GFF3_format() . "\n" if $DEBUG; + $isoform = &trim_isoform($trim_coords_aref, $isoform); + + my $new_cdna_len = $isoform->get_cDNA_length(); + print "\nLengths, before: $cdna_len, targeted: $targeted_cdna_len, now: $new_cdna_len\n" if $DEBUG; + + print "\nAfter: (trimmed cdna length: " . join("-", @$trim_coords_aref) . " of len $cdna_len\n" if $DEBUG; + print "# $isoform_acc trimmed from $cdna_len to $new_cdna_len cdna length\n"; + + $trimmed_flag = 1; + + } + } + + } + + + if ($trimmed_flag) { + $gene_obj_ref->refine_gene_object(); + } + + print $gene_obj_ref->to_GFF3_format() . "\n"; + + + } +} + + +exit(0); + + +#### +sub trim_isoform { + my ($trim_coords_aref, $isoform) = @_; + + my ($trim_end5, $trim_end3) = @$trim_coords_aref; + + my $cdna_length = 0; + + my $orient = $isoform->get_orientation(); + my @exons = $isoform->get_exons(); + + @exons = sort {$a->{end5}<=>$b->{end3}} @exons; + + my @coordsets; + + for (my $i = 0; $i <= $#exons; $i++) { + + my $exon = $exons[$i]; + + my ($end5, $end3) = $exon->get_coords(); + my ($lend, $rend) = sort {$a<=>$b} ($end5, $end3); + + $cdna_length += $rend - $lend + 1; + + if (my $cds = $exon->get_CDS_obj()) { + + my ($cds_end5, $cds_end3) = $cds->get_coords(); + my ($cds_lend, $cds_rend) = sort {$a<=>$b} ($cds_end5, $cds_end3); + + push (@coordsets, [ [$lend,$rend], [$cds_lend,$cds_rend] ]); + } + else { + + push (@coordsets, [ [$lend,$rend], [] ]); + } + } + + + my ($trim_lend, $trim_rend) = ($trim_end5, $trim_end3); + + if ($orient eq '-') { + $trim_lend = $cdna_length - $trim_lend + 1; + $trim_rend = $cdna_length - $trim_rend + 1; + ($trim_lend, $trim_rend) = ($trim_rend, $trim_lend); + } + + + if ($DEBUG) { + print Dumper(\@coordsets); + print "Trimming coordinates: $trim_lend - $trim_rend\n"; + } + + + + ## do trimming. + + my @new_coordsets; + my $cdna_lend_length = 0; + foreach my $coordset (@coordsets) { + my ($exon_coordset, $cds_coordset) = @$coordset; + + my ($exon_lend, $exon_rend) = @$exon_coordset; + my ($cds_lend, $cds_rend) = @$cds_coordset; + + my ($original_exon_lend, $original_exon_rend) = ($exon_lend, $exon_rend); + + my $exon_len = $exon_rend - $exon_lend + 1; + + ## checking Left end: + { + if ((! $cds_lend) && $exon_len + $cdna_lend_length < $trim_lend) { + # utr exon trimmed off + # erase + ($exon_lend, $exon_rend) = (undef, undef); + } + elsif ($trim_lend > $cdna_lend_length && $trim_lend <= $cdna_lend_length + $exon_len) { + ## trim point within exon + my $delta = $trim_lend - $cdna_lend_length; + my $new_exon_lend = $exon_lend + $delta - 1; + $exon_lend = $new_exon_lend; + if ($cds_lend && $exon_lend > $cds_lend) { + $exon_lend = $cds_lend; # just erasing the left utr + } + } + } + + ## checking right end + if ($exon_lend) { + + if ( (! $cds_lend) && $trim_rend < $cdna_lend_length) { + + # trim point is before utr exon + # erase it + ($exon_lend, $exon_rend) = (undef, undef); + } + elsif ($trim_rend > $cdna_lend_length && $trim_rend <= $cdna_lend_length + $exon_len) { + + my $delta = $trim_rend - $cdna_lend_length; + my $new_exon_rend = $original_exon_lend + $delta - 1; + $exon_rend = $new_exon_rend; + + if ($cds_lend && $exon_rend < $cds_rend) { + # just removing right utr + $exon_rend = $cds_rend; + } + } + + } + + push (@new_coordsets, [ [$exon_lend, $exon_rend], [$cds_lend, $cds_rend] ]); + + $cdna_lend_length += $exon_len; + + } + + ## update gene object coordinates. + my %exon_coords; + my %cds_coords; + foreach my $new_coordset (@new_coordsets) { + my ($exon_coords_aref, $cds_coords_aref) = @$new_coordset; + + my ($exon_lend, $exon_rend) = @$exon_coords_aref; + my ($cds_lend, $cds_rend) = @$cds_coords_aref; + + if ($exon_lend) { + my ($exon_end5, $exon_end3) = ($orient eq '+') ? ($exon_lend, $exon_rend) : ($exon_rend, $exon_lend); + $exon_coords{$exon_end5} = $exon_end3; + } + if ($cds_lend) { + my ($cds_end5, $cds_end3) = ($orient eq '+') ? ($cds_lend, $cds_rend) : ($cds_rend, $cds_lend); + $cds_coords{$cds_end5} = $cds_end3; + } + + } + + + $isoform->populate_gene_obj(\%cds_coords, \%exon_coords); + + return($isoform); +} + + + + diff --git a/99.scripts/trinity_utils/util/misc/gff3_to_genome_feature_base_encoding.parse_SAM.pl b/99.scripts/trinity_utils/util/misc/gff3_to_genome_feature_base_encoding.parse_SAM.pl new file mode 100644 index 0000000..abf5d1c --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/gff3_to_genome_feature_base_encoding.parse_SAM.pl @@ -0,0 +1,168 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Gene_obj; +use GFF3_utils; +use SAM_reader; +use SAM_entry; +use Fasta_reader; +use Data::Dumper; +use Getopt::Long qw(:config no_ignore_case bundling); + + +my $usage = <<_EOUSAGE_; + +################################################################### +# +# --encoded_fa feature-encoded fasta file +# +# --coord_sorted_sam sam alignment file +# +# Optional: +# +# --SS_lib_type [FR,RF] (paired), or [R,F] (single) +# +################################################################## + +_EOUSAGE_ + + ; + +my $encoded_fa; +my $sam_file; +my $SS_lib_type; +my $help; + +&GetOptions ( 'h' => \$help, + + 'encoded_fa=s' => \$encoded_fa, + 'coord_sorted_sam=s' => \$sam_file, + + 'SS_lib_type=s' => \$SS_lib_type, + ); + + +if ($help) { + die $usage; +} + +if (! ($encoded_fa && $sam_file) ) { + die $usage; +} + + + +my %SS_encoding = ( 0 => 'intergenic', + 1 => 'intron', + 2 => 'exon+', + 3 => 'exon-', + 4 => 'rRNA+', + 5 => 'rRNA-', + ); + +my %reg_encoding = ( 0 => 'intergenic', + 1 => 'intron', + 2 => 'exon', + 3 => 'exon', + 4 => 'rRNA', + 5 => 'rRNA', + ); + + + +main: { + + + print STDERR "-parsing genome encoding\n"; + my $fasta_reader = new Fasta_reader($encoded_fa); + my %genome_encoding = $fasta_reader->retrieve_all_seqs_hash(); + + my %feature_coverage_counter; + + + print STDERR "-parsing sam file\n"; + + my $curr_scaffold = ""; + my $feat_encoding = ""; + #my @feats; + + my $sam_reader = new SAM_reader($sam_file); + + my $read_counter = 0; + while (my $sam_entry = $sam_reader->get_next()) { + + $read_counter++; + print STDERR "\r[$read_counter] reads processed. " if ($read_counter % 1000 == 0); + + if ($sam_entry->is_query_unmapped()) { + next; + } + + + my $scaffold = $sam_entry->get_scaffold_name(); + if ($scaffold ne $curr_scaffold) { + $curr_scaffold = $scaffold; + $feat_encoding = $genome_encoding{$curr_scaffold} || ""; # no features on a contig are considered intergenic + #@feats = split(//, $feat_encoding); + #print "encoding: $feat_encoding, length: " . length($feat_encoding) . "\n"; + } + + my ($genome_coords_aref, $query_coords_aref) = $sam_entry->get_alignment_coords(); + + my $strand = ($SS_lib_type) ? $sam_entry->get_query_transcribed_strand($SS_lib_type) : $sam_entry->get_query_strand(); + + my $genome_coordset = shift @$genome_coords_aref; + my ($lend, $rend) = @$genome_coordset; + my $midpt = int( ($rend+$lend)/2); + my $encoding = 0; + if ($midpt -1 < length($feat_encoding)) { + $encoding = substr($feat_encoding, $midpt-1, 1); + } + + my $feature_type = ($SS_lib_type) ? $SS_encoding{$encoding} : $reg_encoding{$encoding}; + + if ($SS_lib_type) { + if ($feature_type =~ /([\+\-])$/) { + my $feature_orient = $1; + if ($feature_orient eq $strand && $strand eq '-') { + # -,- + # consider it a forward feature + $feature_type =~ s/\-$/\+/; + } + elsif ($feature_orient ne $strand && $feature_orient eq '+') { + # Feature+,read- : make feature - + $feature_type =~ s/\+$/-/; + } + # Feature-, read+ : counting feature as anti, so already OK + # Feature+, read+ : counting feature as plus, so already OK + } + $feature_coverage_counter{$feature_type}++; + } + else { + ## not strand-specific + $feature_coverage_counter{$feature_type}++; + } + } + + my $total_coverage = 0; + foreach my $count (values %feature_coverage_counter) { + $total_coverage += $count; + } + + + print "\n\n"; + foreach my $feature_type (sort keys %feature_coverage_counter) { + my $coverage = $feature_coverage_counter{$feature_type}; + my $percent = sprintf("%.2f", $coverage/$total_coverage*100); + + print join("\t", $feature_type, $coverage, "$percent\%") . "\n"; + } + print "\n\n"; + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/misc/gff3_to_genome_feature_base_encoding.pl b/99.scripts/trinity_utils/util/misc/gff3_to_genome_feature_base_encoding.pl new file mode 100644 index 0000000..6fc6893 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/gff3_to_genome_feature_base_encoding.pl @@ -0,0 +1,90 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Gene_obj; +use GFF3_utils; +use Data::Dumper; + +my $usage = "usage: $0 file.gff3\n\n"; +my $gff3_file = $ARGV[0] or die $usage; + +my %feat_types = ( intergenic => 0, + intron => 1, + 'exon+' => 2, + 'exon-' => 3, + 'rRNA+' => 4, + 'rRNA-' => 5, + ); + +my $gene_obj_indexer_href = {}; + +## associate gene identifiers with contig id's. +my $contig_to_gene_list_href = &GFF3_utils::index_GFF3_gene_objs($gff3_file, $gene_obj_indexer_href); + +foreach my $asmbl_id (sort keys %$contig_to_gene_list_href) { + + my @pos_vec; + + my @gene_ids = @{$contig_to_gene_list_href->{$asmbl_id}}; + + foreach my $gene_id (@gene_ids) { + my $gene_obj_ref = $gene_obj_indexer_href->{$gene_id}; + + foreach my $isoform ($gene_obj_ref, $gene_obj_ref->get_additional_isoforms()) { + my @introns = $isoform->get_intron_coordinates(); + + foreach my $intron (@introns) { + my ($lend, $rend) = sort {$a<=>$b} @$intron; + + for (my $i = $lend; $i <= $rend; $i++) { + + if ( (! defined $pos_vec[$i]) || $pos_vec[$i] < $feat_types{intron}) { + + $pos_vec[$i] = $feat_types{intron}; + } + } + } + + my $orient = $isoform->get_orientation(); + my $feat_type = "exon$orient"; + if ($isoform->{com_name} =~ /rrna/i && ! $isoform->has_CDS()) { + $feat_type = "rRNA$orient"; + } + + my @exons = $isoform->get_exons(); + foreach my $exon (@exons) { + + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + for (my $i = $lend; $i <= $rend; $i++) { + + if ( (! defined $pos_vec[$i]) || $pos_vec[$i] < $feat_types{$feat_type}) { + + $pos_vec[$i] = $feat_types{$feat_type}; + } + } + } + + } + } + + shift @pos_vec; #rid first position, since we were using 1-based coordinates above. + + foreach my $pos (@pos_vec) { + unless (defined $pos) { + $pos = 0; + } + } + + my $pos_string = join("", @pos_vec); + $pos_string =~ s/(\S{60})/$1\n/g; + + print ">$asmbl_id\n$pos_string\n"; +} + + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/gmap_gff3_chimera_jaccard_analyzer.pl b/99.scripts/trinity_utils/util/misc/gmap_gff3_chimera_jaccard_analyzer.pl new file mode 100644 index 0000000..2a1abfa --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/gmap_gff3_chimera_jaccard_analyzer.pl @@ -0,0 +1,179 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use GFF3_alignment_utils; +use WigParser; +use Data::Dumper; + +my $usage = "usage: $0 gmap.gff3 jaccard.wig [lowest_in_window_size=300]\n\n"; + +my $gmap_gff3 = $ARGV[0] or die $usage; +my $jaccard_wig = $ARGV[1] or die $usage; +my $window_search = $ARGV[2] || 300; + + +main: { + + print STDERR "-indexing wig: $jaccard_wig\n"; + my $wig_parser = new WigParser($jaccard_wig); + + my $alignment_indexer_href = {}; + + + print STDERR "-parsing GFF3 file: $gmap_gff3\n"; + my %contig_to_alignment_ids = &GFF3_alignment_utils::index_GFF3_alignment_objs($gmap_gff3, $alignment_indexer_href); + + ## group multi-paths + my %core_acc_to_alignments; + + + print STDERR "-grouping alignments by acc.\n\n"; + foreach my $contig (keys %contig_to_alignment_ids) { + + my @align_ids = @{$contig_to_alignment_ids{$contig}}; + + foreach my $align_id (@align_ids) { + + my $core_acc = $align_id; + $core_acc =~ s/\.path\d+$//; + + push (@{$core_acc_to_alignments{$core_acc}}, $align_id); + + } + } + + print STDERR "-examining chimeras.\n"; + foreach my $core_acc (keys %core_acc_to_alignments) { + + my @align_ids = @{$core_acc_to_alignments{$core_acc}}; + my $num_alignments = scalar(@align_ids); + if ($num_alignments != 2) { + next; + } + + #print "Got: " . join(", ", @align_ids) . "\n"; + + my @alignment_objs; + my @mcoordsets; + + my @genome_map_entries; + + foreach my $align_id (@align_ids) { + my $align_obj = $alignment_indexer_href->{$align_id} or die "Error, no alignment retrieved for $align_id"; + push (@alignment_objs, $align_obj); + + my @trans_coords = $align_obj->get_mcoords(); + my ($trans_left, $trans_right) = sort {$a<=>$b} @trans_coords; + push (@mcoordsets, [$trans_left, $trans_right]); + + my $genome_acc = $align_obj->{genome_acc}; + + my ($genome_lend, $genome_rend) = $align_obj->get_coords; + push (@genome_map_entries, { + + genome_acc => $genome_acc, + genome_lend => $genome_lend, + genome_rend => $genome_rend, + trans_lend => $trans_coords[0], + trans_rend => $trans_coords[1], + } ); + + } + + @mcoordsets = sort {$a->[0]<=>$b->[0]} @mcoordsets; + + #print "// $core_acc\n" . Dumper(\@mcoordsets) . "\n"; + + + my $clip_pt = $mcoordsets[0]->[1]; + + my $wig_val = -1; + my $num_single = -1; + my $num_both = -1; + eval { + + my @wig_array = $wig_parser->get_wig_array($core_acc, 1); + + if ($#wig_array > $clip_pt) { + $wig_val = &find_lowest_in_window(\@wig_array, $clip_pt, $window_search); #$wig_array[$clip_pt]; + } + if (! defined $wig_val) { + $wig_val = -1; + } + if (ref $wig_val) { + $num_single = $wig_val->[1]; + $num_both = $wig_val->[2]; + $wig_val = $wig_val->[0]; + } + + }; + if ($@) { + print STDERR "No jaccard wig pt for $core_acc\n"; + } + + + + @genome_map_entries = sort {$a->{trans_lend}<=>$b->{trans_lend}} @genome_map_entries; + + my $out_text = join("\t", $core_acc, $clip_pt, $wig_val, $num_single, $num_both); + foreach my $map_entry (@genome_map_entries) { + $out_text .= "\t" . $map_entry->{genome_acc} . ":" + . $map_entry->{genome_lend} . "(" . $map_entry->{trans_lend} . ")" + . "-" + . $map_entry->{genome_rend} . "(" . $map_entry->{trans_rend} . ")"; + } + + print $out_text . "\n"; + + + } + + + exit(0); + + + + + + +} + + +#### +sub find_lowest_in_window { + my ($wig_array_aref, $clip_pt, $window_search) = @_; + + my $left_search = int($clip_pt - ($window_search/2)); + if ($left_search < 1) { $left_search = 1; } + + my $right_search = int($clip_pt + ($window_search/2)); + if ($right_search > $#$wig_array_aref) { + $right_search = $#$wig_array_aref; + } + + my $wig_chosen; + + for (my $i = $left_search; $i <= $right_search; $i++) { + + my $wig_ref = $wig_array_aref->[$i]; + if (ref $wig_ref) { + if ( (! defined $wig_chosen) + || + $wig_ref->[0] < $wig_chosen->[0] + || + ($wig_ref->[0] == $wig_chosen->[0] && $wig_ref->[2] > $wig_chosen->[2]) ) { + + $wig_chosen = $wig_ref; + } + } + } + + return($wig_chosen); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/gmap_gff3_to_percent_length_stats.count_mapped_transcripts.pl b/99.scripts/trinity_utils/util/misc/gmap_gff3_to_percent_length_stats.count_mapped_transcripts.pl new file mode 100644 index 0000000..e444d2e --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/gmap_gff3_to_percent_length_stats.count_mapped_transcripts.pl @@ -0,0 +1,46 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 min_len min_per_len min_per_id file.per_len_and_id ...\n"; + + + +my $min_len = shift @ARGV; +my $min_per_len = shift @ARGV; +my $min_per_id = shift @ARGV; +my @files = @ARGV; + +unless ($min_len && $min_per_len && $min_per_id && @files) { + die $usage; +} + +foreach my $file (@files) { + + my $total_considered = 0; + my $total_OK = 0; + + open (my $fh, $file) or die "Error, cannot open file $file"; + while (<$fh>) { + chomp; + my ($acc, $match_len, $trans_len, $per_len, $per_id) = split(/\t/); + + if ($match_len >= $min_len) { + + $total_considered++; + + if ($per_len >= $min_per_len && $per_id >= $min_per_id) { + + $total_OK++; + } + } + + } + my $percent_OK = sprintf("%.2f", $total_OK / $total_considered); + + print "$file\t$total_considered\t$total_OK\t$percent_OK\n"; +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/gmap_gff3_to_percent_length_stats.pl b/99.scripts/trinity_utils/util/misc/gmap_gff3_to_percent_length_stats.pl new file mode 100644 index 0000000..e802bba --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/gmap_gff3_to_percent_length_stats.pl @@ -0,0 +1,76 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + +my $usage = "usage: $0 gmap.gff3 transcripts.fasta\n\n"; + +my $gmap_gff3 = $ARGV[0] or die $usage; +my $trans_fa = $ARGV[1] or die $usage; + + +main: { + + my %trans_coords_matched; + + open (my $fh, $gmap_gff3) or die $!; + while (<$fh>) { + if (/^\#/) { next; } + chomp; + my @x = split(/\t/); + my $info = $x[8]; + + $info =~ /Target=(\S+) (\d+) (\d+)/ or die "Error, cannot parse transcript and coordinates from $info"; + + my $acc = $1; + my $lend = $2; + my $rend = $3; + my $per_id = $x[5]; + + push (@{$trans_coords_matched{$acc}}, [$lend, $rend, $per_id]); + + } + close $fh; + + + my $fasta_reader = new Fasta_reader($trans_fa); + my %trans_seqs = $fasta_reader->retrieve_all_seqs_hash(); + + foreach my $trans_acc (keys %trans_seqs) { + + my $trans_length = length($trans_seqs{$trans_acc}) or die "Error, no trans seq for $trans_acc"; + + unless (exists $trans_coords_matched{$trans_acc}) { + print join("\t", $trans_acc, 0, $trans_length, 0, 0) . "\n"; + next; + } + + ## compute length and percent identity + my @coords = @{$trans_coords_matched{$trans_acc}}; + + my $match_len = 0; + my $sum_per_id_len = 0; + foreach my $coordset (@coords) { + my ($lend, $rend, $per_id) = @$coordset; + my $len = $rend - $lend + 1; + $match_len += $len; + $sum_per_id_len += $len * $per_id; + } + my $avg_per_id = $sum_per_id_len / $match_len; + + + my $per_length = sprintf("%.2f", $match_len / $trans_length * 100); + + print "$trans_acc\t$match_len\t$trans_length\t$per_length\t$avg_per_id\n"; + } + + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/gmap_native_to_format_converter.pl b/99.scripts/trinity_utils/util/misc/gmap_native_to_format_converter.pl new file mode 100644 index 0000000..c6ad56f --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/gmap_native_to_format_converter.pl @@ -0,0 +1,96 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Gene_obj; + +my $usage = "usage: $0 file.gmap (BED|GTF)\n\n"; + + +my $gmap_file = $ARGV[0] or die $usage; +my $out_format = $ARGV[1] or die $usage; + +unless ($out_format =~ /^(BED|GTF)$/) { + die $usage; +} + + +my %data; + +open (my $fh, $gmap_file) or die $!; +while (<$fh>) { + if (/^>(\S+)/) { + if (%data) { + &format_output(%data); + } + + %data = (); + + my $transcript_acc = $1; + + %data = (acc => $transcript_acc, + segments => {}, # end5 => end3 + ); + + } + elsif (/\s*([\+\-])([^\:]+):(\d+)-(\d+)\s+\((\d+)-(\d+)\)\s+(\d[^\%]+\%)/) { + #print " $transcript_acc\t$_"; + my $orient = $1; + my $genome_acc = $2; + my $genome_end5 = $3; + my $genome_end3 = $4; + my $transcript_end5 = $5; + my $transcript_end3 = $6; + my $percent_identity = $7; + + + $data{segments}->{$genome_end5} = $genome_end3; + $data{scaff} = $genome_acc; + } +} + +if (%data) { + &format_output(%data); +} + +exit(0); + + +#### +sub format_output { + my %data = @_; + + my $gene_obj = new Gene_obj(); + + my $coords_href = $data{segments}; + my $acc = $data{acc}; + my $genome_scaff = $data{scaff}; + + if (%$coords_href) { + + $gene_obj->populate_gene_object($coords_href, $coords_href); + + $gene_obj->{com_name} = $acc; + $gene_obj->{asmbl_id} = $genome_scaff; + $gene_obj->{TU_feat_name} = "g|$acc"; + $gene_obj->{Model_feat_name} = "t|$acc"; + + if ($out_format eq "BED") { + print $gene_obj->to_BED_format(); + } + elsif ($out_format eq "GTF") { + print $gene_obj->to_transcript_GTF_format() . "\n"; + } + else { + die "Error, cannot process format $out_format"; + # shouldn't get here if proper option values processed at top. + } + + + } + + return; +} diff --git a/99.scripts/trinity_utils/util/misc/gtf_to_bed_format.pl b/99.scripts/trinity_utils/util/misc/gtf_to_bed_format.pl new file mode 100644 index 0000000..1d0bc66 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/gtf_to_bed_format.pl @@ -0,0 +1,95 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Gene_obj; + +my $usage = "usage: $0 transcripts.gtf\n\n"; + +my $gtf_file = $ARGV[0] or die $usage; + + +main: { + + my %genome_trans_to_coords; + + open (my $fh, $gtf_file) or die "Error, cannot open file $gtf_file"; + while (<$fh>) { + chomp; + + unless (/\w/) { next; } + + my @x = split(/\t/); + + my $scaff = $x[0]; + my $type = $x[2]; + my $lend = $x[3]; + my $rend = $x[4]; + + my $orient = $x[6]; + + my $info = $x[8]; + + unless ($type eq 'exon') { next; } + + my @parts = split(/;/, $info); + my %atts; + foreach my $part (@parts) { + $part =~ s/^\s+|\s+$//; + $part =~ s/\"//g; + my ($att, $val) = split(/\s+/, $part); + + if (exists $atts{$att}) { + die "Error, already defined attribute $att in $_"; + } + + $atts{$att} = $val; + } + + my $gene_id = $atts{gene_id} or die "Error, no gene_id at $_"; + my $trans_id = $atts{transcript_id} or die "Error, no trans_id at $_"; + + my ($end5, $end3) = ($orient eq '+') ? ($lend, $rend) : ($rend, $lend); + + $genome_trans_to_coords{$scaff}->{$gene_id}->{$trans_id}->{$end5} = $end3; + + } + + + ## Output genes in gff3 format: + + foreach my $scaff (sort keys %genome_trans_to_coords) { + + my $genes_href = $genome_trans_to_coords{$scaff}; + + foreach my $gene_id (keys %$genes_href) { + + my $trans_href = $genes_href->{$gene_id}; + + foreach my $trans_id (keys %$trans_href) { + + my $coords_href = $trans_href->{$trans_id}; + + my $gene_obj = new Gene_obj(); + + $gene_obj->{TU_feat_name} = $gene_id; + $gene_obj->{Model_feat_name} = $trans_id; + $gene_obj->{com_name} = "$gene_id $trans_id"; + + $gene_obj->{asmbl_id} = $scaff; + + $gene_obj->populate_gene_object($coords_href, $coords_href); + + print $gene_obj->to_BED_format(); + + } + } + } + + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/misc/gtf_to_introns.pl b/99.scripts/trinity_utils/util/misc/gtf_to_introns.pl new file mode 100644 index 0000000..aea9bea --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/gtf_to_introns.pl @@ -0,0 +1,125 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Gene_obj; +use Fasta_reader; + +my $usage = "usage: $0 transcripts.gtf genome.fasta\n\n"; + +my $gtf_file = $ARGV[0] or die $usage; +my $genome_fasta_file = $ARGV[1] or die $usage; + +main: { + + + my $fasta_reader = new Fasta_reader($genome_fasta_file); + my %genome_seqs = $fasta_reader->retrieve_all_seqs_hash(); + + + my %genome_trans_to_coords; + + open (my $fh, $gtf_file) or die "Error, cannot open file $gtf_file"; + while (<$fh>) { + chomp; + + unless (/\w/) { next; } + + my @x = split(/\t/); + + my $scaff = $x[0]; + my $type = $x[2]; + my $lend = $x[3]; + my $rend = $x[4]; + + my $orient = $x[6]; + + my $info = $x[8]; + + unless ($type eq 'exon') { next; } + + my @parts = split(/;/, $info); + my %atts; + foreach my $part (@parts) { + $part =~ s/^\s+|\s+$//; + $part =~ s/\"//g; + my ($att, $val) = split(/\s+/, $part); + + if (exists $atts{$att}) { + die "Error, already defined attribute $att in $_"; + } + + $atts{$att} = $val; + } + + my $gene_id = $atts{gene_id} or die "Error, no gene_id at $_"; + my $trans_id = $atts{transcript_id} or die "Error, no trans_id at $_"; + + my ($end5, $end3) = ($orient eq '+') ? ($lend, $rend) : ($rend, $lend); + + $genome_trans_to_coords{$scaff}->{$gene_id}->{$trans_id}->{$end5} = $end3; + + } + + + ## Output genes in gff3 format: + + + + foreach my $scaff (sort keys %genome_trans_to_coords) { + + my $genes_href = $genome_trans_to_coords{$scaff}; + + my $genome_seq = $genome_seqs{$scaff} or die "Error, cannot find genome sequence for acc: $scaff"; + + + foreach my $gene_id (keys %$genes_href) { + + my $trans_href = $genes_href->{$gene_id}; + + my $intron_text = ""; + + foreach my $trans_id (keys %$trans_href) { + + my $coords_href = $trans_href->{$trans_id}; + + my $gene_obj = new Gene_obj(); + + $gene_obj->{TU_feat_name} = $gene_id; + $gene_obj->{Model_feat_name} = $trans_id; + $gene_obj->{com_name} = "$gene_id $trans_id"; + + $gene_obj->{asmbl_id} = $scaff; + + $gene_obj->populate_gene_object($coords_href, $coords_href); + + my $orientation = $gene_obj->get_orientation(); + + #print $gene_obj->to_BED_format(); + + my @intron_coords = $gene_obj->get_intron_coordinates(); + + foreach my $intron (@intron_coords) { + my ($end5, $end3) = @$intron; + my ($lend, $rend) = sort {$a<=>$b} ($end5, $end3); + my $left_splice_dinuc = substr($genome_seq, $lend-1, 2); + my $right_splice_dinuc = substr($genome_seq, $rend-1-1, 2); + + $intron_text .= join("\t", $gene_id, $trans_id, $scaff, "$lend-$rend", $orientation, "$left_splice_dinuc..$right_splice_dinuc") . "\n"; + } + } + + if ($intron_text) { + print "$intron_text\n"; + } + + } + } + + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/misc/hicpipe_raw_converter.pl b/99.scripts/trinity_utils/util/misc/hicpipe_raw_converter.pl new file mode 100644 index 0000000..d45a9d2 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/hicpipe_raw_converter.pl @@ -0,0 +1,23 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 bowtie.raw \n\n"; + +my $bowtie_raw = $ARGV[0] or die $usage; + +open (my $fh, $bowtie_raw) or die "Error, cannot open file $bowtie_raw"; + +my $header = <$fh>; +print $header; + +while (<$fh>) { + if (/random/) { next; } + if (/chrUn/) { next; } + s/chr//g; + print; +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/identify_distal_isoform_variations.pl b/99.scripts/trinity_utils/util/misc/identify_distal_isoform_variations.pl new file mode 100644 index 0000000..e5b1101 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/identify_distal_isoform_variations.pl @@ -0,0 +1,178 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; + +use lib ("$FindBin::RealBin/../../PerlLib", "$FindBin::RealBin/PerlLib"); +use Gene_obj; +use GFF3_utils; + +use SegmentGraph; + +my $usage = "usage: $0 genes.gff3\n\n"; + +my $gff3_file = $ARGV[0] or die $usage; + +main: { + + + my $gene_obj_indexer_href = {}; + ## associate gene identifiers with contig id's. + my $contig_to_gene_list_href = &GFF3_utils::index_GFF3_gene_objs($gff3_file, $gene_obj_indexer_href); + + foreach my $asmbl_id (sort keys %$contig_to_gene_list_href) { + + my @gene_ids = @{$contig_to_gene_list_href->{$asmbl_id}}; + + foreach my $gene_id (@gene_ids) { + my $gene_obj_ref = $gene_obj_indexer_href->{$gene_id}; + + my @additional_isoforms = $gene_obj_ref->get_additional_isoforms(); + + unless (@additional_isoforms) { + next; + } + + my $segment_graph = new SegmentGraph(); + + foreach my $isoform ($gene_obj_ref, @additional_isoforms) { + + my $isoform_id = $isoform->{Model_feat_name}; + + + my @exons = $isoform->get_exons(); + + foreach my $exon (@exons) { + + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + + $segment_graph->add_segment($lend, $rend, $isoform_id); + + } + } + + + #print "Graph for: $gene_id\n"; + #print $segment_graph->toString() . "\n\n\n"; + + &identify_longest_path_between_two_sites_of_variation($gene_id, $segment_graph); + + + + } + + + + } + + exit(0); +} + +#### +sub identify_longest_path_between_two_sites_of_variation { + my ($gene_id, $segment_graph) = @_; + + my @isoforms = $segment_graph->identify_all_owners(); + + print "$gene_id\tISOFORMS: @isoforms\n"; + + for (my $i = 0; $i < $#isoforms; $i++) { + + my $isoform_i = $isoforms[$i]; + + for (my $j = $i + 1; $j <= $#isoforms; $j++) { + + my $isoform_j = $isoforms[$j]; + + my @longest_path = &find_longest_path_between_iso_pair($segment_graph, $isoform_i, $isoform_j); + + + + + + } + + } + + + return; +} + + + +#### +sub find_longest_path_between_iso_pair { + my ($segment_graph, $iso_A, $iso_B) = @_; + + my @nodes = $segment_graph->get_all_nodes(); ## ordered left to right. + + my %seen; + + my @long_path; + my @all_long_paths;; + + + path_search: + foreach my $seed_node (@nodes) { + my $node_ID = $seed_node->get_ID(); + + if ($seen{$node_ID}) { next; } + $seen{$node_ID} = 1; + + unless ($seed_node->has_owners($iso_A, $iso_B)) { + next; + } + + ## ensure left-branched with respect to A,B + + push (@long_path, $seed_node); + my $next_node = $seed_node; + my $forward_OK = 0; + while (1) { + my @next_nodes = $next_node->get_next_nodes(); + unless (@next_nodes) { + last; # reached end w/o finding terminting node. + } + + my $found_next_node = 0; + my $found_A = 0; + my $found_B = 0; + + next_node_candidate_search: + foreach my $next_node_candidate (@next_nodes) { + if ($next_node_candidate->has_owners($iso_A, $iso_B)) { + ## should only be one such path. No cycles allowed in gene structures. + + $found_next_node = 1; + push (@long_path, $next_node); + $next_node = $next_node_candidate; + last next_node_candidate_search; + } + elsif ($next_node_candidate->has_owners($iso_A)) { + $found_A = 1; + } + elsif ($next_node_candidate->has_owners($iso_B)) { + $found_B = 1; + } + } + if (! $found_next_node) { + if ($found_A && $found_B) { + ## excellent, found branched diff. + push (@all_long_paths, [@long_path]); + @long_path = (); + + } + else { + ## doesn't meet our requirements for branched long path diff. + @long_path = (); # clear it out, no save. + + } + last; # break while. + } + } + + } + + +} diff --git a/99.scripts/trinity_utils/util/misc/illustrate_ref_comparison.pl b/99.scripts/trinity_utils/util/misc/illustrate_ref_comparison.pl new file mode 100644 index 0000000..3859903 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/illustrate_ref_comparison.pl @@ -0,0 +1,189 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; + +use lib ("$FindBin::RealBin/../../PerlLib/"); +use Ascii_genome_illustrator; +use Cwd; + +my $usage = "usage: $0 refseqs.fa target.fa min_per_id [min_pct_aligned]\n\n"; + +my $REFSEQS_FA = $ARGV[0] or die $usage; +my $TARGET_FA = $ARGV[1] or die $usage; +my $MIN_PER_ID = $ARGV[2] or die $usage; +my $MIN_PCT_ALIGNED = $ARGV[3] || 0; + +main: { + + + my $cmd = "blat $TARGET_FA $REFSEQS_FA -t=dna -q=dna -out=blast9 $TARGET_FA.blat.out"; + + &process_cmd($cmd); + + # tried blastn, but it doesnt like the Trinity.fasta header format... + #my $cmd = "makeblastdb -in trinity_out_dir/Trinity.fasta -dbtype nucl"; + #$cmd = " blastn -db trinity_out_dir/Trinity.fasta -query refseqs.fa -outfmt 6 > blast.blastn.out"; + #&process_cmd($cmd); + + &generate_ascii_illustration("$TARGET_FA.blat.out"); + + exit(0); +} + +#### +sub generate_ascii_illustration { + my ($align_out_file) = @_; + + my %ref_seq_lengths = &get_seq_lengths($REFSEQS_FA); + my %target_seq_lengths = &get_seq_lengths($TARGET_FA); + + my @hits = &parse_align_out($align_out_file); + + ## Generate a reference view. + foreach my $ref_acc (keys %ref_seq_lengths) { + my $length = $ref_seq_lengths{$ref_acc}; + my $ascii_illustration = new Ascii_genome_illustrator($ref_acc, 60); + + #$ascii_illustration->add_feature($ref_acc, 1, $length, "="); + + my @matches = grep { $_->{ref_acc} eq $ref_acc } @hits; + #@matches = reverse sort {$a->{bitscore}<=>$b->{bitscore}} @matches; + @matches = sort {$a->{ref_start}<=>$b->{ref_start}} @matches; + + + foreach my $match (@matches) { + + my $target_acc = $match->{target_acc}; + my $target_start = $match->{target_start}; + my $target_end = $match->{target_end}; + my $per_id = $match->{per_id}; + + if ($per_id < $MIN_PER_ID) { next; } + + my $ref_start = $match->{ref_start}; + my $ref_end = $match->{ref_end}; + + my $target_length = $target_seq_lengths{$target_acc}; + my $pct_of_target_aligned = sprintf("%.2f", (abs($target_end-$target_start) + 1) / $target_length * 100); + + if ($pct_of_target_aligned < $MIN_PCT_ALIGNED) { next; } + + my $target_feature_name = "$target_acc $target_start-$target_end:$target_length ($pct_of_target_aligned\% aln, $per_id\% ID)"; + + if ($target_start > $target_end) { + ($ref_start, $ref_end) = ($ref_end, $ref_start); + } + + $ascii_illustration->add_feature($target_feature_name, $ref_start, $ref_end, "-"); + + } + + print $ascii_illustration->illustrate(1, $length); + print "\n"; # spacer + + } + + return; +} + +#### +sub parse_align_out { + my ($align_out_file) = @_; + + my @alignments; + + open (my $fh, $align_out_file) or die $!; + while (<$fh>) { + if (/^\#/) { next; } + chomp; + +=blast_outfmt6 + +# Fields: query id, subject id, % identity, alignment length, mismatches, gap opens, q. start, q. end, s. start, s. end, evalue, bit score + + +0 NM_012009_Sh2d1b1 +1 comp0_c0_seq1 +2 100.00 +3 1433 +4 0 +5 0 +6 1 +7 1433 +8 1 +9 1433 +10 0.0 +11 2647 + +=cut + + my @x = split(/\t/); + my ($ref_acc, $target_acc, $per_id, $align_len, $mismatches, $gaps, $ref_start, $ref_end, $target_start, $target_end, $evalue, $bitscore) = @x; + + my $struct = { target_acc => $target_acc, + ref_acc => $ref_acc, + + per_id => $per_id, + mismatches => $mismatches, + gaps => $gaps, + + target_start => $target_start, + target_end => $target_end, + + ref_start => $ref_start, + ref_end => $ref_end, + + evalue => $evalue, + bitscore => $bitscore, + }; + + push (@alignments, $struct); + } + + return(@alignments); + +} + + + + +#### +sub get_seq_lengths { + my ($refseqs_fa) = @_; + + my %lengths; + my $acc; + + open (my $fh, $refseqs_fa) or die $!; + while (<$fh>) { + if (/>(\S+)/) { + $acc = $1; + } + else { + my $seq = $_; + chomp $seq; + my $len = length($seq); + $lengths{$acc} += $len; + } + } + close $fh; + + return(%lengths); +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} diff --git a/99.scripts/trinity_utils/util/misc/insilico_norm_kmer_hists/kmer_histo.NormMaxKCov50.txt b/99.scripts/trinity_utils/util/misc/insilico_norm_kmer_hists/kmer_histo.NormMaxKCov50.txt new file mode 100644 index 0000000..8d4bb4b --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/insilico_norm_kmer_hists/kmer_histo.NormMaxKCov50.txt @@ -0,0 +1,475 @@ +1 199996246 +2 34284052 +3 15833561 +4 10371304 +5 6957144 +6 5350287 +7 3997630 +8 3272664 +9 2613826 +10 2244629 +11 1874971 +12 1655354 +13 1430137 +14 1292231 +15 1158581 +16 1068054 +17 973718 +18 902615 +19 837606 +20 786675 +21 739058 +22 704569 +23 671318 +24 644970 +25 617444 +26 600037 +27 582220 +28 571622 +29 557917 +30 557714 +31 553645 +32 552512 +33 557416 +34 564492 +35 574939 +36 590587 +37 605323 +38 624391 +39 646745 +40 668049 +41 688376 +42 710957 +43 733320 +44 753973 +45 772537 +46 786558 +47 800147 +48 809802 +49 816933 +50 815370 +51 813584 +52 799532 +53 787271 +54 768262 +55 741531 +56 715158 +57 682315 +58 645033 +59 608003 +60 565321 +61 522506 +62 482170 +63 441966 +64 400600 +65 363318 +66 325329 +67 289356 +68 256590 +69 226417 +70 198431 +71 172734 +72 148687 +73 127328 +74 109212 +75 93442 +76 79574 +77 67102 +78 56196 +79 46581 +80 39127 +81 32950 +82 27697 +83 22826 +84 19203 +85 15685 +86 12735 +87 10435 +88 8664 +89 7113 +90 6068 +91 4812 +92 4171 +93 3333 +94 2847 +95 2366 +96 2125 +97 1856 +98 1534 +99 1361 +100 1228 +101 1070 +102 964 +103 810 +104 757 +105 711 +106 643 +107 646 +108 606 +109 579 +110 476 +111 418 +112 464 +113 395 +114 378 +115 376 +116 316 +117 304 +118 317 +119 277 +120 247 +121 243 +122 225 +123 231 +124 199 +125 185 +126 179 +127 195 +128 182 +129 165 +130 137 +131 131 +132 117 +133 128 +134 97 +135 136 +136 103 +137 80 +138 71 +139 67 +140 52 +141 79 +142 54 +143 56 +144 58 +145 44 +146 50 +147 55 +148 51 +149 51 +150 45 +151 64 +152 46 +153 45 +154 37 +155 41 +156 54 +157 36 +158 44 +159 21 +160 34 +161 28 +162 24 +163 17 +164 30 +165 28 +166 31 +167 22 +168 19 +169 23 +170 23 +171 22 +172 17 +173 17 +174 13 +175 16 +176 12 +177 32 +178 22 +179 18 +180 25 +181 17 +182 22 +183 25 +184 15 +185 13 +186 17 +187 17 +188 16 +189 11 +190 10 +191 10 +192 12 +193 8 +194 14 +195 5 +196 15 +197 9 +198 8 +199 7 +200 9 +201 6 +202 7 +203 5 +204 5 +205 6 +206 1 +207 17 +208 9 +209 16 +210 10 +211 3 +212 9 +213 7 +214 7 +215 3 +216 8 +217 6 +218 1 +219 9 +220 3 +221 6 +222 5 +223 6 +224 3 +225 3 +226 7 +227 6 +228 4 +229 6 +230 7 +231 6 +232 6 +233 7 +234 4 +235 7 +236 9 +237 6 +238 2 +239 4 +240 8 +241 1 +242 2 +243 7 +244 4 +245 6 +246 3 +247 3 +248 3 +249 4 +250 5 +251 3 +252 5 +253 6 +254 8 +255 1 +256 2 +257 5 +258 5 +259 5 +260 7 +261 7 +262 6 +263 2 +264 3 +265 5 +266 3 +267 3 +268 3 +269 5 +270 6 +271 2 +272 4 +273 3 +275 3 +276 5 +277 10 +278 1 +279 3 +280 1 +281 6 +282 3 +283 2 +284 1 +285 4 +286 3 +287 1 +288 1 +289 1 +290 4 +291 5 +292 4 +294 2 +295 7 +296 4 +297 1 +298 3 +300 2 +301 4 +302 1 +303 1 +304 5 +305 5 +306 6 +307 2 +308 2 +310 3 +311 4 +312 2 +313 2 +314 3 +315 3 +316 3 +318 2 +320 3 +321 2 +322 1 +323 1 +324 5 +325 4 +326 3 +327 4 +328 3 +329 6 +330 4 +331 4 +332 3 +333 4 +334 1 +335 1 +336 1 +337 1 +338 1 +340 1 +341 2 +345 2 +346 1 +347 1 +348 2 +350 1 +351 2 +352 3 +353 2 +354 4 +355 1 +356 1 +358 3 +360 2 +361 1 +362 1 +364 5 +365 2 +366 1 +367 2 +368 1 +369 1 +371 1 +373 1 +374 2 +378 1 +379 1 +380 2 +381 2 +382 1 +383 1 +386 1 +387 1 +389 1 +390 1 +391 1 +394 2 +395 1 +396 1 +401 2 +402 1 +403 1 +405 2 +407 1 +408 2 +409 1 +412 3 +414 1 +415 1 +416 2 +418 1 +420 1 +423 1 +424 1 +425 1 +441 2 +442 1 +445 2 +453 2 +454 1 +455 1 +456 1 +458 2 +461 1 +470 1 +472 1 +476 1 +477 1 +489 1 +507 1 +509 1 +510 1 +517 1 +518 1 +519 1 +520 2 +521 1 +523 2 +525 1 +532 2 +535 1 +539 1 +547 1 +558 1 +594 1 +624 1 +625 1 +662 1 +670 1 +705 1 +719 1 +784 1 +797 1 +1108 1 +1133 1 +1153 1 +1160 1 +1189 1 +1191 2 +1232 1 +1353 1 +1469 1 +1495 1 +1594 1 +1645 1 +1648 1 +1665 1 +1670 1 +1674 3 +1676 1 +1679 1 +1705 1 +1866 1 +1875 1 +1933 1 +1985 2 +2072 1 +2076 1 +2270 1 +2482 1 +2741 1 +2907 1 +4148 1 +4446 1 +4588 1 +4934 1 +5082 1 +5127 1 +5163 1 +5353 1 +5467 1 +5522 1 +5527 1 +5535 1 +5554 1 +5599 1 +5646 1 +5660 1 +5914 1 +6297 1 +6466 1 +8095 1 diff --git a/99.scripts/trinity_utils/util/misc/insilico_norm_kmer_hists/kmer_histo.all.txt b/99.scripts/trinity_utils/util/misc/insilico_norm_kmer_hists/kmer_histo.all.txt new file mode 100644 index 0000000..79d109f --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/insilico_norm_kmer_hists/kmer_histo.all.txt @@ -0,0 +1,10001 @@ +1 538705521 +2 96187419 +3 44839366 +4 28063291 +5 18509984 +6 13682743 +7 10038145 +8 7950890 +9 6206431 +10 5153883 +11 4195127 +12 3596135 +13 3034333 +14 2668081 +15 2307770 +16 2064751 +17 1826191 +18 1653156 +19 1491501 +20 1360180 +21 1237459 +22 1145817 +23 1056985 +24 980264 +25 910600 +26 854794 +27 796996 +28 754358 +29 706616 +30 672374 +31 634339 +32 600887 +33 570399 +34 545736 +35 521396 +36 498667 +37 475919 +38 458215 +39 439675 +40 424488 +41 405281 +42 392453 +43 377484 +44 365950 +45 355983 +46 342629 +47 331776 +48 319785 +49 311099 +50 302599 +51 294408 +52 285525 +53 276702 +54 268814 +55 262239 +56 255075 +57 248071 +58 242317 +59 237941 +60 230809 +61 225650 +62 221516 +63 216279 +64 210701 +65 206812 +66 202147 +67 197119 +68 194419 +69 190202 +70 186671 +71 182231 +72 177569 +73 175795 +74 171292 +75 169068 +76 165754 +77 161141 +78 158830 +79 155564 +80 153510 +81 150290 +82 148319 +83 145375 +84 142973 +85 140442 +86 137991 +87 135927 +88 133870 +89 131096 +90 129487 +91 128307 +92 126301 +93 124124 +94 122346 +95 120867 +96 118999 +97 117248 +98 115470 +99 114226 +100 112324 +101 110451 +102 109006 +103 107609 +104 105691 +105 104924 +106 103439 +107 101725 +108 101190 +109 99264 +110 98245 +111 96304 +112 95977 +113 94099 +114 93174 +115 91707 +116 90942 +117 89373 +118 88356 +119 87495 +120 86548 +121 85600 +122 84114 +123 83525 +124 82068 +125 81204 +126 79928 +127 79082 +128 78438 +129 77888 +130 76892 +131 76081 +132 75031 +133 74207 +134 73620 +135 72184 +136 71930 +137 70917 +138 69815 +139 69250 +140 69321 +141 67737 +142 67272 +143 67036 +144 65621 +145 65578 +146 64287 +147 63827 +148 63358 +149 62221 +150 62494 +151 60904 +152 60463 +153 60027 +154 59296 +155 58947 +156 58061 +157 57848 +158 57365 +159 56395 +160 55959 +161 55117 +162 54905 +163 53943 +164 54439 +165 53509 +166 52574 +167 52461 +168 52363 +169 51751 +170 51560 +171 51041 +172 50646 +173 49664 +174 49541 +175 49198 +176 48506 +177 48289 +178 47849 +179 47904 +180 47377 +181 46634 +182 46378 +183 45437 +184 45557 +185 45060 +186 44974 +187 44701 +188 44209 +189 43780 +190 43563 +191 43020 +192 42065 +193 42169 +194 41615 +195 41401 +196 40775 +197 40886 +198 40615 +199 40359 +200 40195 +201 39665 +202 39492 +203 38742 +204 38945 +205 38664 +206 38368 +207 37786 +208 37445 +209 37399 +210 37457 +211 36928 +212 36599 +213 36454 +214 36401 +215 35781 +216 35466 +217 35341 +218 35209 +219 35090 +220 35149 +221 34898 +222 34527 +223 34248 +224 34078 +225 33814 +226 33363 +227 32960 +228 33037 +229 32535 +230 32619 +231 32186 +232 31910 +233 31892 +234 31408 +235 31124 +236 31175 +237 30842 +238 30541 +239 30142 +240 30066 +241 30012 +242 29752 +243 29470 +244 29115 +245 29340 +246 28911 +247 28801 +248 28656 +249 28134 +250 28235 +251 27989 +252 27860 +253 27557 +254 27380 +255 27439 +256 27260 +257 27104 +258 26681 +259 26786 +260 26513 +261 26203 +262 26320 +263 25761 +264 25900 +265 25749 +266 25268 +267 25518 +268 25139 +269 25027 +270 24840 +271 24909 +272 24851 +273 24400 +274 24391 +275 23878 +276 24077 +277 23856 +278 23685 +279 23851 +280 23554 +281 23282 +282 23099 +283 22820 +284 23233 +285 22941 +286 22828 +287 22255 +288 22659 +289 22268 +290 22177 +291 21920 +292 21901 +293 21880 +294 21518 +295 21367 +296 21102 +297 21119 +298 20955 +299 21171 +300 20613 +301 20660 +302 20468 +303 20470 +304 20379 +305 20176 +306 19974 +307 20055 +308 19906 +309 19664 +310 19635 +311 19477 +312 19412 +313 19361 +314 19308 +315 19303 +316 19115 +317 19257 +318 18788 +319 18612 +320 18476 +321 18742 +322 18589 +323 18329 +324 18367 +325 18291 +326 17961 +327 17750 +328 17927 +329 17557 +330 17666 +331 17600 +332 17672 +333 17395 +334 17304 +335 17113 +336 17076 +337 17033 +338 16968 +339 16862 +340 16715 +341 16585 +342 16617 +343 16546 +344 16419 +345 16707 +346 16428 +347 16618 +348 16058 +349 16124 +350 16167 +351 15906 +352 15954 +353 15866 +354 15811 +355 15521 +356 15693 +357 15652 +358 15412 +359 15248 +360 15284 +361 15154 +362 15112 +363 14926 +364 15037 +365 14890 +366 14949 +367 14991 +368 14610 +369 14774 +370 14563 +371 14411 +372 14403 +373 14563 +374 14642 +375 14282 +376 14339 +377 14220 +378 14209 +379 14022 +380 13994 +381 13793 +382 13908 +383 13758 +384 13578 +385 13642 +386 13457 +387 13637 +388 13152 +389 13384 +390 13156 +391 13061 +392 13016 +393 12955 +394 13314 +395 13071 +396 12958 +397 13092 +398 12880 +399 12823 +400 12788 +401 12721 +402 12666 +403 12554 +404 12606 +405 12391 +406 12589 +407 12292 +408 12374 +409 12336 +410 12220 +411 12291 +412 12009 +413 12042 +414 12153 +415 11997 +416 11852 +417 11751 +418 11814 +419 11556 +420 11610 +421 11555 +422 11612 +423 11311 +424 11230 +425 11343 +426 11504 +427 11429 +428 11162 +429 11122 +430 11217 +431 11089 +432 11103 +433 11244 +434 11186 +435 10824 +436 10787 +437 10993 +438 10808 +439 10700 +440 10896 +441 10638 +442 10490 +443 10511 +444 10595 +445 10387 +446 10516 +447 10227 +448 10394 +449 10325 +450 10277 +451 10401 +452 10129 +453 10142 +454 10021 +455 9891 +456 9849 +457 9982 +458 9852 +459 9940 +460 9992 +461 9767 +462 9599 +463 9539 +464 9594 +465 9736 +466 9822 +467 9569 +468 9491 +469 9362 +470 9436 +471 9395 +472 9381 +473 9353 +474 9115 +475 9216 +476 9100 +477 9362 +478 9019 +479 9054 +480 9098 +481 9169 +482 8888 +483 9103 +484 8954 +485 8984 +486 9003 +487 8821 +488 8865 +489 8839 +490 8967 +491 8794 +492 8809 +493 8859 +494 8839 +495 8598 +496 8669 +497 8595 +498 8553 +499 8486 +500 8275 +501 8296 +502 8407 +503 8406 +504 8334 +505 8353 +506 8383 +507 8183 +508 8272 +509 8149 +510 8100 +511 8117 +512 8111 +513 7997 +514 8036 +515 8027 +516 7959 +517 7863 +518 7816 +519 7906 +520 7953 +521 7930 +522 7827 +523 7655 +524 7816 +525 7779 +526 7645 +527 7751 +528 7465 +529 7748 +530 7615 +531 7652 +532 7633 +533 7343 +534 7541 +535 7423 +536 7276 +537 7398 +538 7352 +539 7319 +540 7242 +541 7327 +542 7396 +543 7191 +544 7262 +545 7127 +546 7138 +547 7004 +548 7050 +549 7060 +550 7086 +551 6785 +552 6889 +553 6791 +554 6854 +555 6836 +556 6948 +557 6810 +558 6682 +559 6769 +560 6653 +561 6719 +562 6729 +563 6574 +564 6657 +565 6454 +566 6437 +567 6491 +568 6432 +569 6450 +570 6431 +571 6554 +572 6491 +573 6417 +574 6332 +575 6244 +576 6454 +577 6263 +578 6233 +579 6378 +580 6190 +581 6168 +582 6022 +583 6144 +584 6138 +585 6130 +586 6198 +587 6107 +588 6116 +589 6072 +590 6082 +591 6063 +592 6063 +593 5945 +594 5996 +595 5966 +596 5861 +597 5896 +598 5823 +599 5854 +600 6067 +601 5945 +602 5733 +603 5724 +604 5863 +605 5772 +606 5793 +607 5855 +608 5759 +609 5724 +610 5553 +611 5596 +612 5803 +613 5605 +614 5486 +615 5562 +616 5429 +617 5551 +618 5437 +619 5417 +620 5428 +621 5346 +622 5489 +623 5433 +624 5479 +625 5391 +626 5418 +627 5373 +628 5458 +629 5402 +630 5299 +631 5356 +632 5268 +633 5319 +634 5270 +635 5224 +636 5185 +637 5294 +638 5175 +639 5068 +640 5107 +641 5080 +642 5165 +643 5229 +644 5092 +645 5077 +646 5027 +647 5042 +648 4990 +649 4995 +650 5008 +651 5010 +652 4965 +653 5050 +654 4954 +655 5017 +656 4879 +657 4915 +658 4860 +659 4899 +660 4859 +661 4877 +662 4877 +663 4898 +664 4774 +665 4753 +666 4828 +667 4922 +668 4753 +669 4757 +670 4697 +671 4619 +672 4709 +673 4764 +674 4717 +675 4618 +676 4714 +677 4776 +678 4696 +679 4645 +680 4624 +681 4672 +682 4535 +683 4669 +684 4552 +685 4580 +686 4426 +687 4581 +688 4520 +689 4551 +690 4540 +691 4553 +692 4502 +693 4471 +694 4565 +695 4509 +696 4388 +697 4494 +698 4196 +699 4509 +700 4391 +701 4455 +702 4432 +703 4307 +704 4389 +705 4368 +706 4294 +707 4323 +708 4337 +709 4227 +710 4380 +711 4223 +712 4293 +713 4207 +714 4276 +715 4237 +716 4203 +717 4253 +718 4066 +719 4194 +720 4028 +721 4155 +722 4004 +723 4099 +724 3988 +725 3948 +726 3924 +727 3876 +728 4019 +729 3966 +730 3977 +731 3933 +732 3883 +733 3981 +734 3894 +735 3916 +736 3839 +737 3842 +738 3841 +739 4009 +740 3933 +741 3740 +742 3768 +743 3803 +744 3757 +745 3780 +746 3894 +747 3721 +748 3769 +749 3821 +750 3722 +751 3790 +752 3698 +753 3770 +754 3655 +755 3589 +756 3769 +757 3705 +758 3653 +759 3605 +760 3726 +761 3647 +762 3652 +763 3619 +764 3623 +765 3545 +766 3604 +767 3505 +768 3566 +769 3618 +770 3582 +771 3613 +772 3552 +773 3545 +774 3574 +775 3522 +776 3555 +777 3567 +778 3387 +779 3397 +780 3519 +781 3433 +782 3390 +783 3501 +784 3385 +785 3453 +786 3410 +787 3367 +788 3377 +789 3370 +790 3423 +791 3285 +792 3307 +793 3313 +794 3317 +795 3396 +796 3325 +797 3338 +798 3347 +799 3347 +800 3300 +801 3205 +802 3287 +803 3279 +804 3208 +805 3222 +806 3324 +807 3220 +808 3108 +809 3181 +810 3214 +811 3138 +812 3180 +813 3197 +814 3296 +815 3219 +816 3214 +817 3168 +818 3183 +819 3103 +820 3100 +821 3061 +822 3195 +823 3157 +824 3102 +825 3098 +826 3114 +827 3081 +828 2967 +829 3087 +830 3045 +831 3157 +832 3002 +833 3028 +834 3015 +835 3059 +836 3052 +837 3005 +838 3043 +839 3008 +840 3116 +841 3015 +842 3115 +843 3083 +844 3047 +845 3011 +846 2960 +847 2995 +848 2940 +849 3005 +850 2881 +851 2962 +852 3048 +853 2939 +854 2915 +855 2951 +856 2874 +857 2945 +858 2990 +859 2894 +860 2863 +861 2948 +862 2958 +863 2831 +864 2809 +865 2977 +866 2878 +867 2893 +868 2825 +869 2767 +870 2878 +871 2886 +872 2760 +873 2769 +874 2827 +875 2756 +876 2808 +877 2735 +878 2755 +879 2866 +880 2719 +881 2838 +882 2762 +883 2696 +884 2713 +885 2809 +886 2655 +887 2711 +888 2672 +889 2753 +890 2720 +891 2721 +892 2665 +893 2734 +894 2703 +895 2697 +896 2662 +897 2600 +898 2669 +899 2548 +900 2656 +901 2628 +902 2710 +903 2736 +904 2663 +905 2613 +906 2537 +907 2706 +908 2653 +909 2780 +910 2587 +911 2503 +912 2682 +913 2507 +914 2602 +915 2585 +916 2525 +917 2660 +918 2599 +919 2568 +920 2470 +921 2556 +922 2648 +923 2498 +924 2587 +925 2524 +926 2539 +927 2525 +928 2530 +929 2525 +930 2465 +931 2583 +932 2496 +933 2462 +934 2481 +935 2483 +936 2512 +937 2484 +938 2383 +939 2471 +940 2366 +941 2416 +942 2361 +943 2466 +944 2436 +945 2369 +946 2451 +947 2314 +948 2379 +949 2378 +950 2351 +951 2361 +952 2337 +953 2323 +954 2301 +955 2403 +956 2380 +957 2334 +958 2317 +959 2291 +960 2337 +961 2298 +962 2269 +963 2293 +964 2341 +965 2373 +966 2372 +967 2316 +968 2257 +969 2253 +970 2172 +971 2294 +972 2290 +973 2232 +974 2257 +975 2213 +976 2225 +977 2196 +978 2160 +979 2185 +980 2245 +981 2279 +982 2160 +983 2209 +984 2218 +985 2158 +986 2173 +987 2283 +988 2215 +989 2138 +990 2227 +991 2166 +992 2129 +993 2132 +994 2133 +995 2144 +996 2155 +997 2219 +998 2105 +999 2174 +1000 2050 +1001 2060 +1002 2129 +1003 2088 +1004 2032 +1005 2127 +1006 2121 +1007 2110 +1008 2126 +1009 2078 +1010 2066 +1011 2048 +1012 2078 +1013 1990 +1014 2012 +1015 1997 +1016 2016 +1017 2019 +1018 2086 +1019 1997 +1020 2003 +1021 1944 +1022 2031 +1023 1973 +1024 1902 +1025 1993 +1026 1981 +1027 1846 +1028 1967 +1029 1959 +1030 1921 +1031 2008 +1032 1964 +1033 1985 +1034 1939 +1035 1990 +1036 1974 +1037 1901 +1038 1882 +1039 1928 +1040 1917 +1041 1997 +1042 1929 +1043 1900 +1044 1939 +1045 1837 +1046 1834 +1047 1867 +1048 1819 +1049 1856 +1050 1757 +1051 1892 +1052 1864 +1053 1914 +1054 1843 +1055 1884 +1056 1909 +1057 1825 +1058 1838 +1059 1850 +1060 1889 +1061 1863 +1062 1822 +1063 1793 +1064 1871 +1065 1760 +1066 1878 +1067 1828 +1068 1809 +1069 1775 +1070 1757 +1071 1874 +1072 1842 +1073 1838 +1074 1793 +1075 1885 +1076 1802 +1077 1832 +1078 1834 +1079 1845 +1080 1750 +1081 1860 +1082 1848 +1083 1756 +1084 1715 +1085 1820 +1086 1791 +1087 1748 +1088 1828 +1089 1750 +1090 1798 +1091 1822 +1092 1813 +1093 1779 +1094 1781 +1095 1700 +1096 1752 +1097 1751 +1098 1687 +1099 1676 +1100 1780 +1101 1680 +1102 1739 +1103 1737 +1104 1774 +1105 1694 +1106 1657 +1107 1715 +1108 1687 +1109 1626 +1110 1720 +1111 1661 +1112 1650 +1113 1683 +1114 1665 +1115 1660 +1116 1658 +1117 1725 +1118 1680 +1119 1661 +1120 1609 +1121 1636 +1122 1683 +1123 1700 +1124 1667 +1125 1582 +1126 1525 +1127 1615 +1128 1618 +1129 1640 +1130 1664 +1131 1574 +1132 1639 +1133 1621 +1134 1753 +1135 1597 +1136 1547 +1137 1593 +1138 1629 +1139 1566 +1140 1685 +1141 1588 +1142 1644 +1143 1646 +1144 1636 +1145 1630 +1146 1617 +1147 1660 +1148 1556 +1149 1678 +1150 1682 +1151 1584 +1152 1615 +1153 1612 +1154 1591 +1155 1573 +1156 1561 +1157 1531 +1158 1565 +1159 1587 +1160 1542 +1161 1599 +1162 1695 +1163 1551 +1164 1615 +1165 1601 +1166 1545 +1167 1561 +1168 1558 +1169 1608 +1170 1568 +1171 1547 +1172 1595 +1173 1569 +1174 1523 +1175 1485 +1176 1472 +1177 1507 +1178 1563 +1179 1549 +1180 1477 +1181 1576 +1182 1485 +1183 1507 +1184 1559 +1185 1480 +1186 1481 +1187 1567 +1188 1474 +1189 1553 +1190 1522 +1191 1542 +1192 1481 +1193 1496 +1194 1551 +1195 1540 +1196 1490 +1197 1404 +1198 1479 +1199 1506 +1200 1432 +1201 1406 +1202 1470 +1203 1479 +1204 1461 +1205 1448 +1206 1466 +1207 1379 +1208 1516 +1209 1412 +1210 1450 +1211 1439 +1212 1392 +1213 1367 +1214 1486 +1215 1461 +1216 1426 +1217 1413 +1218 1396 +1219 1397 +1220 1473 +1221 1384 +1222 1385 +1223 1415 +1224 1439 +1225 1446 +1226 1397 +1227 1353 +1228 1448 +1229 1353 +1230 1431 +1231 1318 +1232 1430 +1233 1400 +1234 1435 +1235 1288 +1236 1370 +1237 1400 +1238 1445 +1239 1321 +1240 1396 +1241 1384 +1242 1342 +1243 1329 +1244 1380 +1245 1279 +1246 1378 +1247 1399 +1248 1369 +1249 1379 +1250 1307 +1251 1386 +1252 1295 +1253 1296 +1254 1414 +1255 1357 +1256 1398 +1257 1310 +1258 1376 +1259 1351 +1260 1323 +1261 1332 +1262 1377 +1263 1284 +1264 1286 +1265 1382 +1266 1318 +1267 1324 +1268 1359 +1269 1320 +1270 1324 +1271 1318 +1272 1279 +1273 1314 +1274 1359 +1275 1261 +1276 1308 +1277 1403 +1278 1287 +1279 1294 +1280 1244 +1281 1326 +1282 1299 +1283 1284 +1284 1249 +1285 1268 +1286 1282 +1287 1327 +1288 1224 +1289 1236 +1290 1318 +1291 1299 +1292 1176 +1293 1273 +1294 1270 +1295 1206 +1296 1253 +1297 1234 +1298 1238 +1299 1242 +1300 1221 +1301 1176 +1302 1220 +1303 1246 +1304 1299 +1305 1248 +1306 1208 +1307 1252 +1308 1208 +1309 1197 +1310 1234 +1311 1266 +1312 1211 +1313 1246 +1314 1219 +1315 1192 +1316 1265 +1317 1243 +1318 1183 +1319 1154 +1320 1182 +1321 1229 +1322 1209 +1323 1234 +1324 1205 +1325 1171 +1326 1158 +1327 1170 +1328 1136 +1329 1145 +1330 1142 +1331 1227 +1332 1247 +1333 1151 +1334 1215 +1335 1191 +1336 1205 +1337 1132 +1338 1131 +1339 1153 +1340 1195 +1341 1198 +1342 1133 +1343 1227 +1344 1160 +1345 1194 +1346 1116 +1347 1179 +1348 1194 +1349 1139 +1350 1150 +1351 1188 +1352 1140 +1353 1159 +1354 1168 +1355 1099 +1356 1195 +1357 1134 +1358 1166 +1359 1136 +1360 1169 +1361 1093 +1362 1132 +1363 1135 +1364 1162 +1365 1124 +1366 1110 +1367 1174 +1368 1069 +1369 1126 +1370 1131 +1371 1063 +1372 1135 +1373 1143 +1374 1077 +1375 1075 +1376 1145 +1377 1114 +1378 1089 +1379 1060 +1380 1093 +1381 1088 +1382 1131 +1383 1120 +1384 1089 +1385 1077 +1386 1131 +1387 1069 +1388 1070 +1389 1059 +1390 1069 +1391 1062 +1392 1108 +1393 1073 +1394 1084 +1395 1064 +1396 1046 +1397 1107 +1398 1106 +1399 1044 +1400 967 +1401 1085 +1402 1124 +1403 1079 +1404 1089 +1405 1062 +1406 1049 +1407 1071 +1408 1100 +1409 1055 +1410 1088 +1411 1037 +1412 1004 +1413 1030 +1414 1100 +1415 965 +1416 1058 +1417 1036 +1418 1051 +1419 1024 +1420 1037 +1421 1017 +1422 1055 +1423 1077 +1424 1063 +1425 1045 +1426 1028 +1427 1071 +1428 1033 +1429 1043 +1430 1006 +1431 1002 +1432 1069 +1433 1055 +1434 1048 +1435 1022 +1436 1059 +1437 1010 +1438 977 +1439 1030 +1440 974 +1441 1039 +1442 1006 +1443 1018 +1444 965 +1445 1035 +1446 1024 +1447 964 +1448 976 +1449 978 +1450 987 +1451 1039 +1452 988 +1453 934 +1454 974 +1455 980 +1456 984 +1457 1022 +1458 958 +1459 908 +1460 950 +1461 964 +1462 1010 +1463 954 +1464 942 +1465 951 +1466 1002 +1467 962 +1468 954 +1469 957 +1470 979 +1471 978 +1472 965 +1473 987 +1474 959 +1475 917 +1476 957 +1477 980 +1478 951 +1479 946 +1480 970 +1481 927 +1482 1000 +1483 916 +1484 918 +1485 920 +1486 850 +1487 961 +1488 935 +1489 917 +1490 929 +1491 1011 +1492 911 +1493 964 +1494 890 +1495 932 +1496 952 +1497 882 +1498 928 +1499 927 +1500 916 +1501 901 +1502 906 +1503 938 +1504 865 +1505 830 +1506 896 +1507 901 +1508 869 +1509 944 +1510 835 +1511 892 +1512 890 +1513 881 +1514 856 +1515 877 +1516 892 +1517 866 +1518 908 +1519 939 +1520 867 +1521 882 +1522 896 +1523 859 +1524 939 +1525 890 +1526 907 +1527 902 +1528 853 +1529 855 +1530 871 +1531 877 +1532 895 +1533 858 +1534 885 +1535 879 +1536 866 +1537 890 +1538 837 +1539 844 +1540 857 +1541 880 +1542 892 +1543 897 +1544 888 +1545 867 +1546 823 +1547 885 +1548 896 +1549 856 +1550 843 +1551 879 +1552 855 +1553 810 +1554 892 +1555 862 +1556 873 +1557 894 +1558 818 +1559 830 +1560 876 +1561 866 +1562 839 +1563 814 +1564 858 +1565 855 +1566 859 +1567 859 +1568 867 +1569 834 +1570 800 +1571 857 +1572 839 +1573 836 +1574 825 +1575 840 +1576 841 +1577 908 +1578 817 +1579 803 +1580 827 +1581 801 +1582 793 +1583 814 +1584 861 +1585 821 +1586 847 +1587 828 +1588 817 +1589 810 +1590 838 +1591 831 +1592 879 +1593 800 +1594 808 +1595 858 +1596 843 +1597 807 +1598 743 +1599 843 +1600 820 +1601 773 +1602 760 +1603 797 +1604 799 +1605 824 +1606 763 +1607 795 +1608 794 +1609 771 +1610 853 +1611 832 +1612 804 +1613 787 +1614 836 +1615 787 +1616 769 +1617 815 +1618 803 +1619 779 +1620 793 +1621 851 +1622 783 +1623 779 +1624 798 +1625 750 +1626 778 +1627 798 +1628 748 +1629 788 +1630 779 +1631 871 +1632 770 +1633 767 +1634 766 +1635 736 +1636 770 +1637 759 +1638 772 +1639 769 +1640 761 +1641 736 +1642 676 +1643 774 +1644 741 +1645 765 +1646 739 +1647 770 +1648 753 +1649 729 +1650 746 +1651 790 +1652 703 +1653 720 +1654 721 +1655 751 +1656 717 +1657 738 +1658 725 +1659 721 +1660 692 +1661 705 +1662 762 +1663 726 +1664 738 +1665 736 +1666 671 +1667 722 +1668 745 +1669 739 +1670 701 +1671 693 +1672 741 +1673 673 +1674 683 +1675 745 +1676 672 +1677 705 +1678 697 +1679 744 +1680 690 +1681 724 +1682 693 +1683 716 +1684 767 +1685 639 +1686 728 +1687 689 +1688 702 +1689 697 +1690 680 +1691 683 +1692 693 +1693 751 +1694 714 +1695 666 +1696 674 +1697 715 +1698 714 +1699 720 +1700 688 +1701 698 +1702 686 +1703 701 +1704 687 +1705 693 +1706 697 +1707 734 +1708 705 +1709 680 +1710 703 +1711 741 +1712 685 +1713 652 +1714 671 +1715 693 +1716 685 +1717 724 +1718 685 +1719 665 +1720 677 +1721 681 +1722 708 +1723 700 +1724 715 +1725 664 +1726 698 +1727 729 +1728 682 +1729 623 +1730 681 +1731 638 +1732 692 +1733 698 +1734 626 +1735 637 +1736 683 +1737 694 +1738 656 +1739 687 +1740 681 +1741 682 +1742 656 +1743 667 +1744 685 +1745 683 +1746 597 +1747 657 +1748 649 +1749 670 +1750 621 +1751 632 +1752 699 +1753 653 +1754 663 +1755 626 +1756 645 +1757 666 +1758 715 +1759 614 +1760 659 +1761 706 +1762 636 +1763 668 +1764 639 +1765 641 +1766 617 +1767 657 +1768 615 +1769 623 +1770 648 +1771 606 +1772 654 +1773 661 +1774 654 +1775 642 +1776 697 +1777 628 +1778 662 +1779 582 +1780 634 +1781 634 +1782 666 +1783 671 +1784 609 +1785 593 +1786 639 +1787 597 +1788 651 +1789 631 +1790 584 +1791 649 +1792 620 +1793 664 +1794 593 +1795 589 +1796 647 +1797 592 +1798 627 +1799 668 +1800 607 +1801 630 +1802 639 +1803 639 +1804 622 +1805 595 +1806 610 +1807 599 +1808 629 +1809 676 +1810 624 +1811 609 +1812 616 +1813 612 +1814 591 +1815 663 +1816 595 +1817 616 +1818 577 +1819 640 +1820 588 +1821 634 +1822 618 +1823 602 +1824 624 +1825 559 +1826 621 +1827 576 +1828 611 +1829 640 +1830 619 +1831 591 +1832 608 +1833 597 +1834 584 +1835 582 +1836 602 +1837 637 +1838 565 +1839 652 +1840 607 +1841 628 +1842 620 +1843 568 +1844 574 +1845 568 +1846 625 +1847 621 +1848 629 +1849 602 +1850 550 +1851 643 +1852 622 +1853 582 +1854 590 +1855 550 +1856 608 +1857 623 +1858 601 +1859 587 +1860 595 +1861 573 +1862 594 +1863 600 +1864 605 +1865 581 +1866 594 +1867 632 +1868 574 +1869 578 +1870 594 +1871 573 +1872 576 +1873 672 +1874 604 +1875 603 +1876 547 +1877 588 +1878 619 +1879 612 +1880 583 +1881 616 +1882 551 +1883 575 +1884 604 +1885 572 +1886 584 +1887 535 +1888 624 +1889 591 +1890 551 +1891 574 +1892 617 +1893 560 +1894 549 +1895 597 +1896 630 +1897 585 +1898 553 +1899 522 +1900 581 +1901 579 +1902 588 +1903 581 +1904 547 +1905 609 +1906 606 +1907 606 +1908 557 +1909 560 +1910 532 +1911 541 +1912 566 +1913 546 +1914 545 +1915 548 +1916 573 +1917 584 +1918 548 +1919 576 +1920 609 +1921 581 +1922 575 +1923 559 +1924 566 +1925 558 +1926 570 +1927 560 +1928 561 +1929 574 +1930 578 +1931 612 +1932 558 +1933 523 +1934 579 +1935 533 +1936 587 +1937 534 +1938 553 +1939 615 +1940 533 +1941 561 +1942 585 +1943 548 +1944 546 +1945 540 +1946 545 +1947 523 +1948 535 +1949 527 +1950 533 +1951 544 +1952 542 +1953 557 +1954 503 +1955 560 +1956 531 +1957 518 +1958 534 +1959 556 +1960 570 +1961 542 +1962 521 +1963 513 +1964 556 +1965 567 +1966 534 +1967 541 +1968 561 +1969 536 +1970 566 +1971 533 +1972 543 +1973 542 +1974 516 +1975 547 +1976 545 +1977 565 +1978 514 +1979 504 +1980 514 +1981 541 +1982 536 +1983 526 +1984 507 +1985 516 +1986 568 +1987 499 +1988 548 +1989 559 +1990 528 +1991 535 +1992 518 +1993 520 +1994 555 +1995 505 +1996 508 +1997 521 +1998 552 +1999 501 +2000 505 +2001 530 +2002 530 +2003 559 +2004 526 +2005 515 +2006 519 +2007 541 +2008 508 +2009 515 +2010 511 +2011 558 +2012 530 +2013 529 +2014 493 +2015 497 +2016 532 +2017 512 +2018 502 +2019 491 +2020 511 +2021 554 +2022 546 +2023 544 +2024 532 +2025 548 +2026 458 +2027 485 +2028 494 +2029 551 +2030 527 +2031 518 +2032 555 +2033 522 +2034 525 +2035 491 +2036 500 +2037 490 +2038 468 +2039 524 +2040 544 +2041 533 +2042 504 +2043 507 +2044 526 +2045 503 +2046 491 +2047 475 +2048 516 +2049 470 +2050 484 +2051 513 +2052 555 +2053 511 +2054 501 +2055 499 +2056 494 +2057 530 +2058 536 +2059 507 +2060 490 +2061 515 +2062 503 +2063 484 +2064 468 +2065 471 +2066 485 +2067 500 +2068 425 +2069 459 +2070 435 +2071 507 +2072 476 +2073 450 +2074 524 +2075 472 +2076 503 +2077 478 +2078 507 +2079 493 +2080 484 +2081 497 +2082 499 +2083 488 +2084 472 +2085 488 +2086 491 +2087 494 +2088 484 +2089 492 +2090 468 +2091 474 +2092 445 +2093 520 +2094 452 +2095 454 +2096 474 +2097 508 +2098 445 +2099 466 +2100 508 +2101 463 +2102 401 +2103 481 +2104 475 +2105 477 +2106 442 +2107 475 +2108 448 +2109 438 +2110 455 +2111 463 +2112 478 +2113 462 +2114 446 +2115 409 +2116 467 +2117 469 +2118 440 +2119 446 +2120 466 +2121 484 +2122 493 +2123 443 +2124 467 +2125 454 +2126 448 +2127 448 +2128 453 +2129 472 +2130 463 +2131 475 +2132 440 +2133 450 +2134 438 +2135 417 +2136 464 +2137 473 +2138 451 +2139 475 +2140 456 +2141 433 +2142 428 +2143 449 +2144 473 +2145 453 +2146 430 +2147 417 +2148 458 +2149 453 +2150 451 +2151 437 +2152 426 +2153 450 +2154 429 +2155 456 +2156 418 +2157 418 +2158 473 +2159 456 +2160 433 +2161 456 +2162 445 +2163 417 +2164 420 +2165 475 +2166 433 +2167 432 +2168 389 +2169 461 +2170 461 +2171 450 +2172 446 +2173 447 +2174 410 +2175 451 +2176 419 +2177 401 +2178 399 +2179 427 +2180 387 +2181 467 +2182 426 +2183 435 +2184 457 +2185 448 +2186 411 +2187 456 +2188 457 +2189 472 +2190 474 +2191 425 +2192 404 +2193 412 +2194 426 +2195 472 +2196 419 +2197 440 +2198 385 +2199 402 +2200 413 +2201 460 +2202 403 +2203 430 +2204 402 +2205 437 +2206 430 +2207 441 +2208 401 +2209 418 +2210 412 +2211 455 +2212 445 +2213 411 +2214 415 +2215 440 +2216 457 +2217 412 +2218 419 +2219 406 +2220 419 +2221 420 +2222 420 +2223 407 +2224 449 +2225 411 +2226 397 +2227 444 +2228 387 +2229 395 +2230 402 +2231 409 +2232 396 +2233 416 +2234 418 +2235 395 +2236 395 +2237 389 +2238 445 +2239 392 +2240 371 +2241 438 +2242 421 +2243 416 +2244 424 +2245 399 +2246 420 +2247 407 +2248 381 +2249 444 +2250 419 +2251 404 +2252 423 +2253 408 +2254 373 +2255 376 +2256 374 +2257 462 +2258 380 +2259 441 +2260 417 +2261 429 +2262 378 +2263 402 +2264 414 +2265 411 +2266 405 +2267 401 +2268 392 +2269 400 +2270 425 +2271 378 +2272 370 +2273 446 +2274 400 +2275 442 +2276 410 +2277 395 +2278 388 +2279 378 +2280 412 +2281 390 +2282 374 +2283 387 +2284 352 +2285 415 +2286 351 +2287 359 +2288 417 +2289 398 +2290 381 +2291 390 +2292 366 +2293 402 +2294 404 +2295 404 +2296 413 +2297 384 +2298 375 +2299 392 +2300 391 +2301 363 +2302 396 +2303 405 +2304 384 +2305 398 +2306 368 +2307 385 +2308 403 +2309 410 +2310 404 +2311 349 +2312 391 +2313 357 +2314 408 +2315 415 +2316 371 +2317 395 +2318 402 +2319 410 +2320 404 +2321 387 +2322 359 +2323 351 +2324 381 +2325 384 +2326 410 +2327 364 +2328 432 +2329 404 +2330 394 +2331 385 +2332 367 +2333 385 +2334 393 +2335 439 +2336 387 +2337 407 +2338 382 +2339 389 +2340 406 +2341 418 +2342 373 +2343 388 +2344 379 +2345 366 +2346 392 +2347 381 +2348 409 +2349 389 +2350 397 +2351 394 +2352 397 +2353 428 +2354 388 +2355 365 +2356 393 +2357 353 +2358 358 +2359 321 +2360 369 +2361 353 +2362 362 +2363 372 +2364 374 +2365 364 +2366 343 +2367 391 +2368 387 +2369 382 +2370 386 +2371 369 +2372 390 +2373 364 +2374 370 +2375 401 +2376 378 +2377 367 +2378 337 +2379 362 +2380 377 +2381 343 +2382 358 +2383 381 +2384 404 +2385 383 +2386 366 +2387 345 +2388 388 +2389 339 +2390 369 +2391 352 +2392 407 +2393 350 +2394 357 +2395 356 +2396 372 +2397 372 +2398 370 +2399 331 +2400 381 +2401 348 +2402 350 +2403 362 +2404 347 +2405 351 +2406 340 +2407 354 +2408 346 +2409 353 +2410 396 +2411 367 +2412 351 +2413 397 +2414 340 +2415 386 +2416 333 +2417 349 +2418 330 +2419 347 +2420 368 +2421 371 +2422 329 +2423 372 +2424 319 +2425 366 +2426 332 +2427 383 +2428 341 +2429 376 +2430 334 +2431 361 +2432 364 +2433 375 +2434 351 +2435 341 +2436 338 +2437 350 +2438 386 +2439 336 +2440 368 +2441 331 +2442 343 +2443 334 +2444 356 +2445 359 +2446 312 +2447 344 +2448 352 +2449 314 +2450 318 +2451 329 +2452 327 +2453 334 +2454 334 +2455 349 +2456 345 +2457 311 +2458 303 +2459 333 +2460 314 +2461 344 +2462 346 +2463 342 +2464 359 +2465 356 +2466 344 +2467 347 +2468 348 +2469 299 +2470 370 +2471 302 +2472 343 +2473 331 +2474 335 +2475 344 +2476 323 +2477 316 +2478 345 +2479 324 +2480 363 +2481 325 +2482 354 +2483 337 +2484 281 +2485 348 +2486 350 +2487 345 +2488 360 +2489 324 +2490 352 +2491 318 +2492 287 +2493 344 +2494 308 +2495 319 +2496 351 +2497 325 +2498 313 +2499 325 +2500 325 +2501 313 +2502 333 +2503 318 +2504 349 +2505 361 +2506 315 +2507 343 +2508 322 +2509 354 +2510 326 +2511 309 +2512 346 +2513 341 +2514 322 +2515 331 +2516 335 +2517 343 +2518 348 +2519 296 +2520 342 +2521 341 +2522 312 +2523 347 +2524 377 +2525 323 +2526 320 +2527 337 +2528 306 +2529 315 +2530 338 +2531 328 +2532 336 +2533 331 +2534 313 +2535 291 +2536 336 +2537 315 +2538 313 +2539 326 +2540 318 +2541 303 +2542 324 +2543 312 +2544 338 +2545 306 +2546 312 +2547 323 +2548 297 +2549 331 +2550 323 +2551 314 +2552 323 +2553 331 +2554 287 +2555 312 +2556 309 +2557 347 +2558 333 +2559 314 +2560 312 +2561 310 +2562 316 +2563 306 +2564 322 +2565 332 +2566 329 +2567 324 +2568 347 +2569 332 +2570 312 +2571 324 +2572 308 +2573 289 +2574 335 +2575 312 +2576 321 +2577 305 +2578 300 +2579 331 +2580 311 +2581 308 +2582 298 +2583 317 +2584 298 +2585 318 +2586 313 +2587 293 +2588 291 +2589 310 +2590 310 +2591 304 +2592 298 +2593 272 +2594 322 +2595 311 +2596 295 +2597 297 +2598 298 +2599 300 +2600 313 +2601 297 +2602 306 +2603 276 +2604 291 +2605 309 +2606 295 +2607 276 +2608 302 +2609 289 +2610 335 +2611 250 +2612 295 +2613 297 +2614 282 +2615 283 +2616 304 +2617 283 +2618 262 +2619 295 +2620 271 +2621 289 +2622 287 +2623 291 +2624 288 +2625 305 +2626 263 +2627 310 +2628 261 +2629 283 +2630 301 +2631 283 +2632 307 +2633 300 +2634 304 +2635 308 +2636 306 +2637 287 +2638 272 +2639 266 +2640 283 +2641 267 +2642 281 +2643 279 +2644 252 +2645 295 +2646 278 +2647 255 +2648 276 +2649 269 +2650 291 +2651 288 +2652 270 +2653 270 +2654 267 +2655 294 +2656 269 +2657 283 +2658 277 +2659 300 +2660 281 +2661 287 +2662 274 +2663 303 +2664 282 +2665 278 +2666 296 +2667 267 +2668 267 +2669 268 +2670 263 +2671 286 +2672 299 +2673 287 +2674 250 +2675 262 +2676 262 +2677 262 +2678 279 +2679 259 +2680 313 +2681 290 +2682 296 +2683 262 +2684 296 +2685 284 +2686 284 +2687 246 +2688 317 +2689 310 +2690 260 +2691 295 +2692 270 +2693 258 +2694 278 +2695 249 +2696 285 +2697 287 +2698 292 +2699 250 +2700 302 +2701 306 +2702 285 +2703 258 +2704 264 +2705 290 +2706 288 +2707 281 +2708 254 +2709 259 +2710 256 +2711 286 +2712 253 +2713 275 +2714 286 +2715 295 +2716 306 +2717 259 +2718 292 +2719 237 +2720 273 +2721 245 +2722 253 +2723 258 +2724 293 +2725 266 +2726 253 +2727 267 +2728 266 +2729 261 +2730 276 +2731 288 +2732 268 +2733 248 +2734 267 +2735 273 +2736 242 +2737 271 +2738 243 +2739 266 +2740 232 +2741 251 +2742 271 +2743 278 +2744 280 +2745 264 +2746 260 +2747 235 +2748 276 +2749 265 +2750 257 +2751 269 +2752 266 +2753 272 +2754 239 +2755 282 +2756 269 +2757 269 +2758 238 +2759 254 +2760 274 +2761 273 +2762 249 +2763 253 +2764 268 +2765 274 +2766 261 +2767 268 +2768 254 +2769 268 +2770 241 +2771 260 +2772 271 +2773 268 +2774 252 +2775 218 +2776 276 +2777 271 +2778 283 +2779 271 +2780 258 +2781 263 +2782 274 +2783 250 +2784 278 +2785 259 +2786 250 +2787 260 +2788 248 +2789 266 +2790 261 +2791 252 +2792 247 +2793 246 +2794 255 +2795 288 +2796 244 +2797 261 +2798 257 +2799 258 +2800 273 +2801 291 +2802 222 +2803 250 +2804 266 +2805 227 +2806 217 +2807 256 +2808 258 +2809 251 +2810 268 +2811 247 +2812 262 +2813 245 +2814 227 +2815 294 +2816 235 +2817 223 +2818 237 +2819 241 +2820 262 +2821 257 +2822 240 +2823 245 +2824 221 +2825 282 +2826 206 +2827 225 +2828 216 +2829 262 +2830 230 +2831 266 +2832 246 +2833 228 +2834 228 +2835 240 +2836 254 +2837 221 +2838 227 +2839 258 +2840 263 +2841 238 +2842 231 +2843 220 +2844 233 +2845 256 +2846 272 +2847 248 +2848 228 +2849 229 +2850 243 +2851 211 +2852 234 +2853 235 +2854 268 +2855 235 +2856 220 +2857 237 +2858 234 +2859 242 +2860 216 +2861 223 +2862 247 +2863 228 +2864 220 +2865 226 +2866 227 +2867 248 +2868 238 +2869 234 +2870 245 +2871 261 +2872 264 +2873 262 +2874 224 +2875 223 +2876 226 +2877 274 +2878 238 +2879 236 +2880 244 +2881 258 +2882 265 +2883 209 +2884 232 +2885 244 +2886 266 +2887 253 +2888 215 +2889 224 +2890 219 +2891 250 +2892 235 +2893 251 +2894 223 +2895 208 +2896 246 +2897 226 +2898 235 +2899 232 +2900 221 +2901 221 +2902 245 +2903 249 +2904 237 +2905 205 +2906 241 +2907 232 +2908 246 +2909 239 +2910 248 +2911 232 +2912 258 +2913 274 +2914 230 +2915 218 +2916 242 +2917 216 +2918 231 +2919 212 +2920 243 +2921 242 +2922 247 +2923 254 +2924 246 +2925 240 +2926 213 +2927 196 +2928 234 +2929 224 +2930 238 +2931 243 +2932 225 +2933 223 +2934 203 +2935 240 +2936 246 +2937 228 +2938 235 +2939 209 +2940 248 +2941 216 +2942 261 +2943 233 +2944 219 +2945 210 +2946 211 +2947 234 +2948 230 +2949 234 +2950 189 +2951 217 +2952 256 +2953 222 +2954 251 +2955 228 +2956 236 +2957 209 +2958 198 +2959 220 +2960 240 +2961 218 +2962 212 +2963 209 +2964 225 +2965 228 +2966 232 +2967 232 +2968 221 +2969 216 +2970 217 +2971 245 +2972 208 +2973 236 +2974 221 +2975 211 +2976 240 +2977 215 +2978 237 +2979 232 +2980 219 +2981 216 +2982 232 +2983 233 +2984 202 +2985 213 +2986 207 +2987 209 +2988 202 +2989 214 +2990 205 +2991 225 +2992 206 +2993 224 +2994 229 +2995 244 +2996 208 +2997 244 +2998 225 +2999 238 +3000 207 +3001 222 +3002 220 +3003 219 +3004 204 +3005 266 +3006 223 +3007 213 +3008 212 +3009 220 +3010 197 +3011 210 +3012 209 +3013 215 +3014 214 +3015 216 +3016 228 +3017 233 +3018 226 +3019 213 +3020 214 +3021 219 +3022 226 +3023 217 +3024 234 +3025 256 +3026 252 +3027 229 +3028 210 +3029 218 +3030 217 +3031 211 +3032 245 +3033 226 +3034 230 +3035 221 +3036 227 +3037 229 +3038 189 +3039 208 +3040 231 +3041 206 +3042 207 +3043 223 +3044 218 +3045 224 +3046 238 +3047 208 +3048 210 +3049 205 +3050 224 +3051 205 +3052 207 +3053 224 +3054 230 +3055 187 +3056 204 +3057 204 +3058 202 +3059 186 +3060 210 +3061 190 +3062 216 +3063 198 +3064 240 +3065 225 +3066 241 +3067 200 +3068 207 +3069 224 +3070 202 +3071 217 +3072 211 +3073 205 +3074 211 +3075 187 +3076 186 +3077 215 +3078 193 +3079 220 +3080 211 +3081 192 +3082 197 +3083 223 +3084 205 +3085 222 +3086 212 +3087 214 +3088 176 +3089 201 +3090 213 +3091 198 +3092 190 +3093 234 +3094 235 +3095 224 +3096 211 +3097 220 +3098 189 +3099 220 +3100 196 +3101 217 +3102 220 +3103 208 +3104 210 +3105 234 +3106 193 +3107 202 +3108 191 +3109 208 +3110 212 +3111 225 +3112 185 +3113 215 +3114 210 +3115 228 +3116 212 +3117 229 +3118 200 +3119 196 +3120 191 +3121 206 +3122 188 +3123 208 +3124 224 +3125 211 +3126 183 +3127 221 +3128 221 +3129 210 +3130 215 +3131 217 +3132 227 +3133 202 +3134 185 +3135 200 +3136 193 +3137 205 +3138 208 +3139 197 +3140 207 +3141 204 +3142 206 +3143 195 +3144 227 +3145 221 +3146 210 +3147 222 +3148 194 +3149 202 +3150 164 +3151 190 +3152 232 +3153 196 +3154 192 +3155 213 +3156 187 +3157 201 +3158 207 +3159 221 +3160 179 +3161 218 +3162 215 +3163 194 +3164 198 +3165 204 +3166 202 +3167 182 +3168 227 +3169 199 +3170 213 +3171 188 +3172 210 +3173 205 +3174 162 +3175 211 +3176 191 +3177 179 +3178 185 +3179 188 +3180 188 +3181 201 +3182 205 +3183 208 +3184 185 +3185 208 +3186 187 +3187 185 +3188 175 +3189 197 +3190 203 +3191 179 +3192 190 +3193 174 +3194 194 +3195 222 +3196 198 +3197 191 +3198 218 +3199 229 +3200 182 +3201 194 +3202 184 +3203 214 +3204 198 +3205 171 +3206 173 +3207 183 +3208 192 +3209 198 +3210 191 +3211 179 +3212 161 +3213 178 +3214 220 +3215 196 +3216 191 +3217 204 +3218 210 +3219 212 +3220 176 +3221 208 +3222 173 +3223 191 +3224 201 +3225 163 +3226 175 +3227 211 +3228 173 +3229 189 +3230 179 +3231 177 +3232 197 +3233 186 +3234 199 +3235 183 +3236 197 +3237 175 +3238 194 +3239 210 +3240 190 +3241 187 +3242 214 +3243 189 +3244 169 +3245 169 +3246 173 +3247 196 +3248 199 +3249 205 +3250 162 +3251 195 +3252 194 +3253 179 +3254 180 +3255 167 +3256 170 +3257 171 +3258 185 +3259 177 +3260 158 +3261 171 +3262 195 +3263 174 +3264 180 +3265 174 +3266 173 +3267 201 +3268 183 +3269 182 +3270 177 +3271 156 +3272 172 +3273 160 +3274 196 +3275 184 +3276 202 +3277 212 +3278 148 +3279 172 +3280 178 +3281 160 +3282 185 +3283 204 +3284 154 +3285 185 +3286 176 +3287 200 +3288 163 +3289 189 +3290 174 +3291 191 +3292 177 +3293 200 +3294 181 +3295 188 +3296 152 +3297 162 +3298 177 +3299 190 +3300 171 +3301 179 +3302 157 +3303 175 +3304 161 +3305 184 +3306 161 +3307 213 +3308 170 +3309 210 +3310 164 +3311 186 +3312 201 +3313 192 +3314 177 +3315 195 +3316 145 +3317 184 +3318 178 +3319 184 +3320 176 +3321 182 +3322 185 +3323 196 +3324 178 +3325 181 +3326 180 +3327 169 +3328 178 +3329 213 +3330 157 +3331 177 +3332 172 +3333 176 +3334 193 +3335 169 +3336 183 +3337 175 +3338 153 +3339 158 +3340 180 +3341 185 +3342 208 +3343 178 +3344 177 +3345 192 +3346 184 +3347 181 +3348 159 +3349 182 +3350 165 +3351 182 +3352 166 +3353 149 +3354 165 +3355 186 +3356 168 +3357 186 +3358 172 +3359 178 +3360 171 +3361 167 +3362 160 +3363 183 +3364 144 +3365 148 +3366 164 +3367 179 +3368 163 +3369 181 +3370 180 +3371 167 +3372 181 +3373 147 +3374 162 +3375 169 +3376 192 +3377 176 +3378 191 +3379 169 +3380 156 +3381 172 +3382 163 +3383 168 +3384 154 +3385 179 +3386 170 +3387 182 +3388 148 +3389 187 +3390 182 +3391 167 +3392 172 +3393 189 +3394 183 +3395 164 +3396 184 +3397 168 +3398 155 +3399 166 +3400 165 +3401 176 +3402 179 +3403 164 +3404 161 +3405 173 +3406 171 +3407 156 +3408 179 +3409 164 +3410 168 +3411 174 +3412 159 +3413 197 +3414 168 +3415 189 +3416 181 +3417 150 +3418 158 +3419 158 +3420 196 +3421 173 +3422 168 +3423 160 +3424 169 +3425 169 +3426 179 +3427 152 +3428 175 +3429 149 +3430 193 +3431 175 +3432 158 +3433 154 +3434 167 +3435 171 +3436 158 +3437 157 +3438 164 +3439 179 +3440 154 +3441 170 +3442 166 +3443 159 +3444 137 +3445 153 +3446 162 +3447 161 +3448 170 +3449 159 +3450 154 +3451 161 +3452 146 +3453 167 +3454 167 +3455 158 +3456 178 +3457 157 +3458 154 +3459 178 +3460 175 +3461 175 +3462 153 +3463 160 +3464 177 +3465 137 +3466 164 +3467 158 +3468 150 +3469 155 +3470 135 +3471 152 +3472 146 +3473 141 +3474 184 +3475 165 +3476 162 +3477 160 +3478 161 +3479 170 +3480 167 +3481 160 +3482 155 +3483 151 +3484 141 +3485 163 +3486 160 +3487 152 +3488 145 +3489 149 +3490 143 +3491 155 +3492 187 +3493 173 +3494 168 +3495 158 +3496 153 +3497 163 +3498 165 +3499 168 +3500 156 +3501 170 +3502 157 +3503 177 +3504 163 +3505 162 +3506 144 +3507 153 +3508 161 +3509 168 +3510 142 +3511 155 +3512 186 +3513 143 +3514 147 +3515 148 +3516 190 +3517 142 +3518 142 +3519 167 +3520 179 +3521 172 +3522 139 +3523 156 +3524 144 +3525 158 +3526 172 +3527 165 +3528 169 +3529 155 +3530 145 +3531 161 +3532 154 +3533 150 +3534 156 +3535 149 +3536 156 +3537 133 +3538 148 +3539 160 +3540 152 +3541 147 +3542 158 +3543 151 +3544 183 +3545 149 +3546 143 +3547 152 +3548 158 +3549 153 +3550 140 +3551 153 +3552 156 +3553 158 +3554 170 +3555 138 +3556 174 +3557 183 +3558 158 +3559 153 +3560 143 +3561 147 +3562 139 +3563 159 +3564 127 +3565 152 +3566 171 +3567 159 +3568 160 +3569 155 +3570 175 +3571 167 +3572 156 +3573 171 +3574 146 +3575 120 +3576 148 +3577 165 +3578 154 +3579 150 +3580 162 +3581 166 +3582 155 +3583 159 +3584 153 +3585 139 +3586 157 +3587 123 +3588 150 +3589 141 +3590 155 +3591 131 +3592 151 +3593 169 +3594 145 +3595 171 +3596 149 +3597 159 +3598 157 +3599 159 +3600 161 +3601 154 +3602 120 +3603 158 +3604 145 +3605 144 +3606 150 +3607 162 +3608 150 +3609 139 +3610 144 +3611 149 +3612 147 +3613 161 +3614 151 +3615 147 +3616 143 +3617 162 +3618 149 +3619 141 +3620 126 +3621 144 +3622 153 +3623 169 +3624 151 +3625 166 +3626 148 +3627 130 +3628 142 +3629 140 +3630 152 +3631 170 +3632 190 +3633 137 +3634 165 +3635 160 +3636 154 +3637 133 +3638 134 +3639 136 +3640 131 +3641 158 +3642 128 +3643 163 +3644 125 +3645 140 +3646 148 +3647 140 +3648 158 +3649 156 +3650 153 +3651 141 +3652 154 +3653 156 +3654 138 +3655 146 +3656 146 +3657 147 +3658 159 +3659 135 +3660 137 +3661 167 +3662 122 +3663 146 +3664 144 +3665 156 +3666 175 +3667 145 +3668 146 +3669 143 +3670 163 +3671 128 +3672 162 +3673 148 +3674 166 +3675 150 +3676 140 +3677 141 +3678 144 +3679 123 +3680 138 +3681 121 +3682 148 +3683 150 +3684 152 +3685 166 +3686 151 +3687 136 +3688 115 +3689 141 +3690 137 +3691 148 +3692 136 +3693 151 +3694 127 +3695 143 +3696 153 +3697 150 +3698 129 +3699 142 +3700 145 +3701 122 +3702 149 +3703 137 +3704 119 +3705 136 +3706 152 +3707 150 +3708 149 +3709 156 +3710 156 +3711 145 +3712 138 +3713 131 +3714 148 +3715 139 +3716 171 +3717 143 +3718 137 +3719 156 +3720 142 +3721 144 +3722 145 +3723 148 +3724 128 +3725 137 +3726 169 +3727 136 +3728 131 +3729 147 +3730 110 +3731 159 +3732 140 +3733 149 +3734 141 +3735 144 +3736 128 +3737 137 +3738 117 +3739 145 +3740 128 +3741 127 +3742 145 +3743 129 +3744 136 +3745 135 +3746 134 +3747 126 +3748 148 +3749 114 +3750 161 +3751 146 +3752 125 +3753 127 +3754 130 +3755 129 +3756 144 +3757 126 +3758 141 +3759 121 +3760 145 +3761 138 +3762 131 +3763 131 +3764 149 +3765 149 +3766 139 +3767 123 +3768 144 +3769 148 +3770 137 +3771 148 +3772 135 +3773 129 +3774 122 +3775 141 +3776 121 +3777 119 +3778 121 +3779 110 +3780 160 +3781 129 +3782 135 +3783 116 +3784 125 +3785 142 +3786 119 +3787 136 +3788 129 +3789 130 +3790 145 +3791 136 +3792 134 +3793 139 +3794 119 +3795 132 +3796 134 +3797 112 +3798 127 +3799 137 +3800 135 +3801 120 +3802 123 +3803 139 +3804 117 +3805 121 +3806 129 +3807 141 +3808 130 +3809 145 +3810 111 +3811 146 +3812 123 +3813 159 +3814 137 +3815 131 +3816 143 +3817 131 +3818 128 +3819 132 +3820 124 +3821 120 +3822 117 +3823 117 +3824 124 +3825 139 +3826 138 +3827 136 +3828 137 +3829 144 +3830 116 +3831 127 +3832 118 +3833 144 +3834 120 +3835 121 +3836 142 +3837 119 +3838 121 +3839 124 +3840 136 +3841 110 +3842 144 +3843 117 +3844 146 +3845 136 +3846 148 +3847 142 +3848 131 +3849 139 +3850 128 +3851 129 +3852 115 +3853 129 +3854 133 +3855 123 +3856 126 +3857 154 +3858 117 +3859 121 +3860 144 +3861 122 +3862 140 +3863 122 +3864 145 +3865 109 +3866 139 +3867 114 +3868 117 +3869 132 +3870 130 +3871 132 +3872 133 +3873 137 +3874 121 +3875 111 +3876 126 +3877 143 +3878 129 +3879 140 +3880 115 +3881 140 +3882 154 +3883 141 +3884 131 +3885 136 +3886 126 +3887 102 +3888 127 +3889 132 +3890 125 +3891 116 +3892 133 +3893 144 +3894 142 +3895 146 +3896 124 +3897 131 +3898 134 +3899 127 +3900 126 +3901 124 +3902 113 +3903 115 +3904 112 +3905 114 +3906 116 +3907 134 +3908 143 +3909 130 +3910 124 +3911 122 +3912 125 +3913 118 +3914 118 +3915 129 +3916 124 +3917 134 +3918 125 +3919 108 +3920 130 +3921 146 +3922 141 +3923 116 +3924 124 +3925 119 +3926 142 +3927 147 +3928 135 +3929 125 +3930 127 +3931 139 +3932 113 +3933 135 +3934 130 +3935 117 +3936 134 +3937 121 +3938 133 +3939 108 +3940 130 +3941 116 +3942 139 +3943 121 +3944 113 +3945 124 +3946 130 +3947 138 +3948 122 +3949 120 +3950 132 +3951 125 +3952 109 +3953 118 +3954 122 +3955 126 +3956 111 +3957 125 +3958 136 +3959 131 +3960 105 +3961 139 +3962 145 +3963 114 +3964 158 +3965 123 +3966 122 +3967 125 +3968 126 +3969 129 +3970 123 +3971 119 +3972 121 +3973 124 +3974 131 +3975 123 +3976 125 +3977 135 +3978 139 +3979 129 +3980 138 +3981 127 +3982 110 +3983 118 +3984 120 +3985 146 +3986 118 +3987 117 +3988 148 +3989 118 +3990 119 +3991 131 +3992 114 +3993 105 +3994 125 +3995 117 +3996 114 +3997 130 +3998 97 +3999 126 +4000 127 +4001 102 +4002 138 +4003 121 +4004 101 +4005 124 +4006 126 +4007 101 +4008 115 +4009 117 +4010 97 +4011 96 +4012 124 +4013 126 +4014 136 +4015 133 +4016 105 +4017 130 +4018 127 +4019 138 +4020 110 +4021 129 +4022 121 +4023 147 +4024 122 +4025 113 +4026 106 +4027 126 +4028 107 +4029 122 +4030 117 +4031 121 +4032 119 +4033 128 +4034 95 +4035 96 +4036 113 +4037 104 +4038 123 +4039 98 +4040 113 +4041 110 +4042 127 +4043 127 +4044 122 +4045 111 +4046 111 +4047 121 +4048 119 +4049 114 +4050 120 +4051 119 +4052 137 +4053 105 +4054 91 +4055 96 +4056 113 +4057 113 +4058 122 +4059 115 +4060 128 +4061 118 +4062 115 +4063 121 +4064 124 +4065 127 +4066 117 +4067 114 +4068 122 +4069 135 +4070 130 +4071 119 +4072 120 +4073 118 +4074 109 +4075 120 +4076 103 +4077 101 +4078 137 +4079 112 +4080 101 +4081 110 +4082 125 +4083 134 +4084 110 +4085 112 +4086 132 +4087 116 +4088 116 +4089 133 +4090 102 +4091 136 +4092 115 +4093 114 +4094 96 +4095 145 +4096 100 +4097 136 +4098 118 +4099 126 +4100 105 +4101 118 +4102 123 +4103 137 +4104 113 +4105 100 +4106 100 +4107 100 +4108 105 +4109 121 +4110 114 +4111 119 +4112 98 +4113 127 +4114 131 +4115 114 +4116 114 +4117 118 +4118 118 +4119 128 +4120 130 +4121 102 +4122 116 +4123 118 +4124 105 +4125 102 +4126 120 +4127 111 +4128 116 +4129 114 +4130 113 +4131 114 +4132 129 +4133 126 +4134 106 +4135 126 +4136 120 +4137 127 +4138 104 +4139 130 +4140 103 +4141 113 +4142 114 +4143 113 +4144 120 +4145 105 +4146 117 +4147 114 +4148 116 +4149 119 +4150 126 +4151 108 +4152 104 +4153 123 +4154 118 +4155 147 +4156 107 +4157 112 +4158 122 +4159 126 +4160 130 +4161 96 +4162 110 +4163 121 +4164 110 +4165 98 +4166 99 +4167 125 +4168 118 +4169 119 +4170 129 +4171 118 +4172 91 +4173 113 +4174 98 +4175 115 +4176 120 +4177 104 +4178 104 +4179 122 +4180 100 +4181 104 +4182 119 +4183 121 +4184 122 +4185 107 +4186 122 +4187 110 +4188 119 +4189 125 +4190 113 +4191 120 +4192 101 +4193 105 +4194 110 +4195 104 +4196 88 +4197 93 +4198 108 +4199 124 +4200 117 +4201 122 +4202 119 +4203 106 +4204 99 +4205 106 +4206 116 +4207 121 +4208 109 +4209 114 +4210 113 +4211 104 +4212 127 +4213 107 +4214 92 +4215 112 +4216 106 +4217 116 +4218 107 +4219 117 +4220 109 +4221 119 +4222 104 +4223 100 +4224 106 +4225 131 +4226 113 +4227 109 +4228 111 +4229 97 +4230 112 +4231 105 +4232 103 +4233 122 +4234 98 +4235 119 +4236 104 +4237 103 +4238 128 +4239 119 +4240 98 +4241 124 +4242 98 +4243 109 +4244 107 +4245 113 +4246 117 +4247 95 +4248 125 +4249 103 +4250 104 +4251 94 +4252 111 +4253 95 +4254 111 +4255 124 +4256 107 +4257 86 +4258 95 +4259 131 +4260 98 +4261 121 +4262 100 +4263 103 +4264 114 +4265 106 +4266 124 +4267 109 +4268 113 +4269 111 +4270 110 +4271 112 +4272 102 +4273 98 +4274 128 +4275 109 +4276 93 +4277 107 +4278 117 +4279 103 +4280 92 +4281 104 +4282 96 +4283 106 +4284 115 +4285 98 +4286 112 +4287 111 +4288 107 +4289 95 +4290 108 +4291 110 +4292 85 +4293 113 +4294 109 +4295 109 +4296 96 +4297 111 +4298 100 +4299 112 +4300 111 +4301 110 +4302 113 +4303 127 +4304 111 +4305 127 +4306 95 +4307 103 +4308 94 +4309 113 +4310 95 +4311 88 +4312 101 +4313 126 +4314 106 +4315 92 +4316 122 +4317 97 +4318 115 +4319 86 +4320 108 +4321 111 +4322 80 +4323 103 +4324 100 +4325 101 +4326 100 +4327 94 +4328 87 +4329 118 +4330 106 +4331 116 +4332 99 +4333 105 +4334 107 +4335 111 +4336 103 +4337 114 +4338 104 +4339 96 +4340 102 +4341 113 +4342 95 +4343 100 +4344 113 +4345 101 +4346 115 +4347 112 +4348 101 +4349 109 +4350 108 +4351 102 +4352 107 +4353 82 +4354 96 +4355 97 +4356 110 +4357 104 +4358 99 +4359 112 +4360 100 +4361 97 +4362 106 +4363 81 +4364 105 +4365 90 +4366 103 +4367 112 +4368 114 +4369 106 +4370 80 +4371 105 +4372 129 +4373 117 +4374 97 +4375 82 +4376 98 +4377 104 +4378 88 +4379 99 +4380 115 +4381 89 +4382 99 +4383 101 +4384 108 +4385 106 +4386 97 +4387 106 +4388 113 +4389 93 +4390 112 +4391 112 +4392 111 +4393 111 +4394 103 +4395 106 +4396 116 +4397 127 +4398 89 +4399 107 +4400 94 +4401 113 +4402 95 +4403 119 +4404 100 +4405 120 +4406 89 +4407 123 +4408 114 +4409 119 +4410 116 +4411 99 +4412 104 +4413 92 +4414 119 +4415 106 +4416 114 +4417 98 +4418 101 +4419 89 +4420 119 +4421 105 +4422 96 +4423 108 +4424 91 +4425 105 +4426 104 +4427 95 +4428 109 +4429 121 +4430 99 +4431 95 +4432 87 +4433 102 +4434 114 +4435 99 +4436 95 +4437 122 +4438 97 +4439 87 +4440 97 +4441 98 +4442 104 +4443 87 +4444 90 +4445 103 +4446 79 +4447 106 +4448 131 +4449 102 +4450 113 +4451 103 +4452 91 +4453 78 +4454 128 +4455 110 +4456 89 +4457 90 +4458 106 +4459 91 +4460 101 +4461 109 +4462 104 +4463 97 +4464 101 +4465 94 +4466 106 +4467 100 +4468 118 +4469 106 +4470 86 +4471 97 +4472 93 +4473 84 +4474 105 +4475 99 +4476 88 +4477 104 +4478 114 +4479 109 +4480 91 +4481 91 +4482 119 +4483 88 +4484 111 +4485 109 +4486 94 +4487 117 +4488 100 +4489 102 +4490 82 +4491 96 +4492 100 +4493 104 +4494 95 +4495 99 +4496 102 +4497 101 +4498 117 +4499 107 +4500 97 +4501 85 +4502 90 +4503 106 +4504 109 +4505 90 +4506 85 +4507 97 +4508 82 +4509 107 +4510 109 +4511 103 +4512 89 +4513 88 +4514 94 +4515 100 +4516 126 +4517 100 +4518 111 +4519 97 +4520 98 +4521 98 +4522 84 +4523 92 +4524 100 +4525 98 +4526 86 +4527 93 +4528 108 +4529 98 +4530 96 +4531 93 +4532 98 +4533 99 +4534 97 +4535 93 +4536 109 +4537 90 +4538 84 +4539 82 +4540 93 +4541 91 +4542 92 +4543 96 +4544 75 +4545 100 +4546 98 +4547 83 +4548 111 +4549 98 +4550 86 +4551 82 +4552 93 +4553 114 +4554 92 +4555 98 +4556 106 +4557 95 +4558 91 +4559 100 +4560 106 +4561 85 +4562 104 +4563 96 +4564 88 +4565 78 +4566 87 +4567 107 +4568 98 +4569 103 +4570 86 +4571 92 +4572 109 +4573 104 +4574 112 +4575 94 +4576 93 +4577 88 +4578 88 +4579 89 +4580 90 +4581 111 +4582 101 +4583 108 +4584 92 +4585 91 +4586 93 +4587 91 +4588 98 +4589 104 +4590 97 +4591 79 +4592 114 +4593 87 +4594 101 +4595 101 +4596 85 +4597 92 +4598 111 +4599 109 +4600 109 +4601 94 +4602 109 +4603 101 +4604 88 +4605 78 +4606 92 +4607 88 +4608 86 +4609 94 +4610 93 +4611 89 +4612 99 +4613 99 +4614 103 +4615 101 +4616 92 +4617 99 +4618 99 +4619 81 +4620 109 +4621 101 +4622 95 +4623 94 +4624 76 +4625 94 +4626 84 +4627 103 +4628 90 +4629 107 +4630 109 +4631 93 +4632 95 +4633 112 +4634 102 +4635 106 +4636 88 +4637 78 +4638 79 +4639 87 +4640 109 +4641 93 +4642 101 +4643 96 +4644 96 +4645 118 +4646 93 +4647 83 +4648 102 +4649 94 +4650 91 +4651 104 +4652 76 +4653 95 +4654 90 +4655 113 +4656 96 +4657 95 +4658 98 +4659 103 +4660 96 +4661 87 +4662 87 +4663 110 +4664 98 +4665 97 +4666 94 +4667 93 +4668 80 +4669 112 +4670 100 +4671 88 +4672 91 +4673 110 +4674 90 +4675 90 +4676 87 +4677 93 +4678 91 +4679 109 +4680 87 +4681 107 +4682 74 +4683 105 +4684 96 +4685 96 +4686 82 +4687 85 +4688 105 +4689 80 +4690 91 +4691 86 +4692 66 +4693 105 +4694 102 +4695 77 +4696 83 +4697 87 +4698 91 +4699 96 +4700 83 +4701 86 +4702 87 +4703 93 +4704 98 +4705 86 +4706 89 +4707 83 +4708 85 +4709 95 +4710 112 +4711 85 +4712 78 +4713 91 +4714 99 +4715 99 +4716 89 +4717 97 +4718 99 +4719 92 +4720 84 +4721 92 +4722 92 +4723 86 +4724 97 +4725 99 +4726 96 +4727 96 +4728 90 +4729 89 +4730 95 +4731 98 +4732 101 +4733 89 +4734 88 +4735 87 +4736 90 +4737 81 +4738 107 +4739 110 +4740 92 +4741 88 +4742 86 +4743 84 +4744 96 +4745 90 +4746 83 +4747 84 +4748 96 +4749 93 +4750 89 +4751 70 +4752 84 +4753 84 +4754 83 +4755 109 +4756 81 +4757 87 +4758 89 +4759 89 +4760 101 +4761 92 +4762 84 +4763 92 +4764 104 +4765 78 +4766 98 +4767 86 +4768 90 +4769 94 +4770 89 +4771 102 +4772 83 +4773 89 +4774 77 +4775 78 +4776 87 +4777 94 +4778 92 +4779 76 +4780 70 +4781 91 +4782 72 +4783 99 +4784 82 +4785 80 +4786 72 +4787 88 +4788 96 +4789 84 +4790 97 +4791 84 +4792 80 +4793 86 +4794 89 +4795 84 +4796 81 +4797 86 +4798 91 +4799 92 +4800 87 +4801 80 +4802 85 +4803 98 +4804 70 +4805 81 +4806 82 +4807 91 +4808 84 +4809 105 +4810 94 +4811 77 +4812 97 +4813 87 +4814 71 +4815 85 +4816 81 +4817 88 +4818 90 +4819 79 +4820 97 +4821 72 +4822 90 +4823 74 +4824 98 +4825 81 +4826 95 +4827 71 +4828 88 +4829 99 +4830 86 +4831 92 +4832 87 +4833 82 +4834 88 +4835 71 +4836 78 +4837 87 +4838 75 +4839 86 +4840 87 +4841 87 +4842 71 +4843 80 +4844 92 +4845 84 +4846 97 +4847 84 +4848 103 +4849 76 +4850 94 +4851 80 +4852 74 +4853 91 +4854 75 +4855 65 +4856 85 +4857 103 +4858 91 +4859 72 +4860 67 +4861 100 +4862 72 +4863 77 +4864 79 +4865 103 +4866 80 +4867 82 +4868 79 +4869 87 +4870 78 +4871 75 +4872 79 +4873 90 +4874 82 +4875 91 +4876 88 +4877 81 +4878 93 +4879 82 +4880 76 +4881 94 +4882 83 +4883 86 +4884 93 +4885 80 +4886 92 +4887 84 +4888 94 +4889 80 +4890 76 +4891 86 +4892 87 +4893 86 +4894 83 +4895 71 +4896 84 +4897 80 +4898 87 +4899 81 +4900 75 +4901 79 +4902 81 +4903 102 +4904 84 +4905 84 +4906 78 +4907 88 +4908 90 +4909 83 +4910 69 +4911 68 +4912 87 +4913 84 +4914 90 +4915 90 +4916 86 +4917 86 +4918 91 +4919 81 +4920 84 +4921 87 +4922 91 +4923 77 +4924 97 +4925 84 +4926 83 +4927 93 +4928 80 +4929 78 +4930 93 +4931 85 +4932 99 +4933 84 +4934 86 +4935 97 +4936 102 +4937 78 +4938 78 +4939 75 +4940 79 +4941 92 +4942 77 +4943 89 +4944 82 +4945 98 +4946 91 +4947 85 +4948 98 +4949 76 +4950 76 +4951 97 +4952 79 +4953 86 +4954 80 +4955 63 +4956 76 +4957 80 +4958 85 +4959 93 +4960 85 +4961 81 +4962 92 +4963 71 +4964 89 +4965 87 +4966 91 +4967 74 +4968 84 +4969 74 +4970 98 +4971 100 +4972 91 +4973 81 +4974 84 +4975 75 +4976 88 +4977 70 +4978 89 +4979 82 +4980 95 +4981 71 +4982 80 +4983 82 +4984 83 +4985 81 +4986 83 +4987 87 +4988 78 +4989 79 +4990 78 +4991 96 +4992 90 +4993 102 +4994 94 +4995 85 +4996 98 +4997 86 +4998 71 +4999 76 +5000 93 +5001 85 +5002 77 +5003 85 +5004 96 +5005 72 +5006 93 +5007 85 +5008 85 +5009 92 +5010 87 +5011 85 +5012 90 +5013 98 +5014 70 +5015 87 +5016 80 +5017 89 +5018 72 +5019 84 +5020 94 +5021 73 +5022 94 +5023 72 +5024 76 +5025 86 +5026 79 +5027 92 +5028 79 +5029 95 +5030 81 +5031 81 +5032 97 +5033 76 +5034 75 +5035 77 +5036 63 +5037 82 +5038 88 +5039 98 +5040 84 +5041 71 +5042 85 +5043 94 +5044 90 +5045 92 +5046 95 +5047 78 +5048 78 +5049 73 +5050 83 +5051 85 +5052 80 +5053 82 +5054 72 +5055 68 +5056 84 +5057 68 +5058 81 +5059 88 +5060 71 +5061 86 +5062 102 +5063 89 +5064 67 +5065 84 +5066 89 +5067 78 +5068 74 +5069 71 +5070 80 +5071 63 +5072 76 +5073 63 +5074 71 +5075 88 +5076 95 +5077 92 +5078 88 +5079 67 +5080 74 +5081 75 +5082 77 +5083 72 +5084 69 +5085 76 +5086 90 +5087 74 +5088 63 +5089 91 +5090 74 +5091 67 +5092 79 +5093 68 +5094 88 +5095 83 +5096 69 +5097 68 +5098 78 +5099 91 +5100 97 +5101 84 +5102 84 +5103 59 +5104 95 +5105 86 +5106 79 +5107 85 +5108 83 +5109 86 +5110 73 +5111 73 +5112 79 +5113 84 +5114 62 +5115 74 +5116 76 +5117 76 +5118 73 +5119 77 +5120 90 +5121 88 +5122 78 +5123 69 +5124 93 +5125 83 +5126 86 +5127 84 +5128 74 +5129 74 +5130 88 +5131 81 +5132 81 +5133 74 +5134 84 +5135 84 +5136 67 +5137 82 +5138 74 +5139 86 +5140 90 +5141 84 +5142 83 +5143 79 +5144 68 +5145 84 +5146 72 +5147 75 +5148 72 +5149 92 +5150 83 +5151 76 +5152 83 +5153 71 +5154 66 +5155 73 +5156 73 +5157 83 +5158 95 +5159 69 +5160 83 +5161 69 +5162 83 +5163 60 +5164 73 +5165 68 +5166 75 +5167 67 +5168 78 +5169 75 +5170 77 +5171 89 +5172 64 +5173 78 +5174 63 +5175 77 +5176 77 +5177 88 +5178 86 +5179 74 +5180 72 +5181 74 +5182 83 +5183 83 +5184 75 +5185 78 +5186 80 +5187 86 +5188 74 +5189 75 +5190 80 +5191 72 +5192 67 +5193 76 +5194 79 +5195 78 +5196 70 +5197 75 +5198 78 +5199 67 +5200 71 +5201 84 +5202 71 +5203 65 +5204 75 +5205 68 +5206 80 +5207 85 +5208 84 +5209 87 +5210 68 +5211 78 +5212 75 +5213 72 +5214 84 +5215 65 +5216 84 +5217 85 +5218 60 +5219 63 +5220 77 +5221 79 +5222 67 +5223 73 +5224 69 +5225 87 +5226 78 +5227 71 +5228 71 +5229 77 +5230 72 +5231 81 +5232 69 +5233 61 +5234 62 +5235 67 +5236 79 +5237 77 +5238 89 +5239 83 +5240 68 +5241 67 +5242 69 +5243 60 +5244 65 +5245 73 +5246 75 +5247 80 +5248 79 +5249 79 +5250 68 +5251 72 +5252 80 +5253 65 +5254 85 +5255 80 +5256 73 +5257 79 +5258 72 +5259 62 +5260 86 +5261 74 +5262 67 +5263 78 +5264 73 +5265 71 +5266 66 +5267 93 +5268 63 +5269 80 +5270 74 +5271 86 +5272 75 +5273 93 +5274 78 +5275 86 +5276 78 +5277 92 +5278 84 +5279 66 +5280 72 +5281 82 +5282 65 +5283 66 +5284 71 +5285 64 +5286 69 +5287 82 +5288 85 +5289 75 +5290 83 +5291 83 +5292 74 +5293 74 +5294 76 +5295 81 +5296 77 +5297 75 +5298 73 +5299 84 +5300 74 +5301 72 +5302 71 +5303 81 +5304 69 +5305 62 +5306 82 +5307 72 +5308 70 +5309 88 +5310 66 +5311 67 +5312 78 +5313 71 +5314 86 +5315 77 +5316 69 +5317 70 +5318 87 +5319 60 +5320 84 +5321 74 +5322 88 +5323 84 +5324 79 +5325 73 +5326 77 +5327 72 +5328 72 +5329 76 +5330 82 +5331 71 +5332 58 +5333 85 +5334 72 +5335 83 +5336 66 +5337 76 +5338 65 +5339 74 +5340 74 +5341 77 +5342 94 +5343 77 +5344 74 +5345 79 +5346 80 +5347 87 +5348 69 +5349 75 +5350 78 +5351 58 +5352 70 +5353 89 +5354 74 +5355 60 +5356 74 +5357 66 +5358 60 +5359 66 +5360 78 +5361 70 +5362 79 +5363 58 +5364 85 +5365 87 +5366 77 +5367 70 +5368 66 +5369 68 +5370 78 +5371 69 +5372 65 +5373 72 +5374 76 +5375 82 +5376 90 +5377 62 +5378 78 +5379 67 +5380 89 +5381 71 +5382 73 +5383 80 +5384 104 +5385 64 +5386 81 +5387 73 +5388 68 +5389 81 +5390 67 +5391 82 +5392 72 +5393 77 +5394 69 +5395 73 +5396 76 +5397 59 +5398 80 +5399 63 +5400 74 +5401 70 +5402 90 +5403 68 +5404 62 +5405 67 +5406 59 +5407 68 +5408 79 +5409 59 +5410 67 +5411 69 +5412 66 +5413 58 +5414 86 +5415 76 +5416 58 +5417 69 +5418 85 +5419 55 +5420 81 +5421 57 +5422 78 +5423 56 +5424 69 +5425 71 +5426 60 +5427 71 +5428 74 +5429 65 +5430 59 +5431 82 +5432 71 +5433 85 +5434 60 +5435 63 +5436 62 +5437 63 +5438 71 +5439 64 +5440 71 +5441 87 +5442 69 +5443 70 +5444 54 +5445 55 +5446 66 +5447 73 +5448 87 +5449 60 +5450 65 +5451 68 +5452 66 +5453 67 +5454 72 +5455 55 +5456 67 +5457 73 +5458 73 +5459 55 +5460 73 +5461 69 +5462 61 +5463 68 +5464 68 +5465 65 +5466 78 +5467 81 +5468 63 +5469 74 +5470 74 +5471 78 +5472 81 +5473 62 +5474 65 +5475 68 +5476 70 +5477 67 +5478 68 +5479 64 +5480 61 +5481 65 +5482 65 +5483 72 +5484 55 +5485 70 +5486 59 +5487 73 +5488 63 +5489 78 +5490 57 +5491 61 +5492 86 +5493 68 +5494 75 +5495 80 +5496 70 +5497 72 +5498 65 +5499 65 +5500 70 +5501 86 +5502 69 +5503 76 +5504 56 +5505 63 +5506 64 +5507 70 +5508 56 +5509 77 +5510 51 +5511 55 +5512 69 +5513 61 +5514 54 +5515 79 +5516 73 +5517 68 +5518 71 +5519 74 +5520 68 +5521 78 +5522 68 +5523 74 +5524 53 +5525 64 +5526 67 +5527 63 +5528 71 +5529 62 +5530 63 +5531 67 +5532 82 +5533 58 +5534 71 +5535 78 +5536 59 +5537 70 +5538 62 +5539 63 +5540 67 +5541 48 +5542 62 +5543 60 +5544 58 +5545 82 +5546 65 +5547 70 +5548 75 +5549 73 +5550 61 +5551 67 +5552 80 +5553 68 +5554 62 +5555 76 +5556 68 +5557 52 +5558 68 +5559 60 +5560 87 +5561 53 +5562 95 +5563 63 +5564 53 +5565 79 +5566 55 +5567 54 +5568 67 +5569 63 +5570 65 +5571 77 +5572 49 +5573 66 +5574 61 +5575 64 +5576 60 +5577 51 +5578 52 +5579 71 +5580 67 +5581 62 +5582 61 +5583 67 +5584 76 +5585 66 +5586 69 +5587 68 +5588 66 +5589 48 +5590 67 +5591 60 +5592 64 +5593 66 +5594 57 +5595 88 +5596 71 +5597 55 +5598 55 +5599 65 +5600 65 +5601 103 +5602 70 +5603 64 +5604 63 +5605 47 +5606 60 +5607 66 +5608 67 +5609 65 +5610 54 +5611 65 +5612 57 +5613 64 +5614 67 +5615 65 +5616 56 +5617 52 +5618 66 +5619 51 +5620 68 +5621 81 +5622 64 +5623 68 +5624 53 +5625 61 +5626 70 +5627 62 +5628 51 +5629 69 +5630 67 +5631 70 +5632 67 +5633 72 +5634 61 +5635 74 +5636 71 +5637 62 +5638 68 +5639 66 +5640 49 +5641 46 +5642 58 +5643 65 +5644 74 +5645 72 +5646 71 +5647 81 +5648 58 +5649 67 +5650 68 +5651 60 +5652 49 +5653 62 +5654 61 +5655 60 +5656 61 +5657 61 +5658 67 +5659 59 +5660 65 +5661 64 +5662 65 +5663 71 +5664 61 +5665 73 +5666 67 +5667 71 +5668 66 +5669 60 +5670 61 +5671 53 +5672 57 +5673 73 +5674 69 +5675 63 +5676 63 +5677 73 +5678 71 +5679 66 +5680 72 +5681 62 +5682 65 +5683 59 +5684 66 +5685 57 +5686 67 +5687 54 +5688 62 +5689 56 +5690 56 +5691 66 +5692 46 +5693 74 +5694 50 +5695 59 +5696 67 +5697 61 +5698 58 +5699 65 +5700 55 +5701 52 +5702 98 +5703 64 +5704 57 +5705 61 +5706 49 +5707 55 +5708 67 +5709 61 +5710 48 +5711 68 +5712 68 +5713 63 +5714 52 +5715 77 +5716 63 +5717 56 +5718 63 +5719 50 +5720 60 +5721 70 +5722 58 +5723 73 +5724 58 +5725 56 +5726 59 +5727 59 +5728 72 +5729 68 +5730 67 +5731 67 +5732 78 +5733 59 +5734 65 +5735 51 +5736 72 +5737 56 +5738 68 +5739 74 +5740 60 +5741 70 +5742 67 +5743 57 +5744 84 +5745 66 +5746 57 +5747 57 +5748 80 +5749 63 +5750 61 +5751 63 +5752 60 +5753 64 +5754 52 +5755 58 +5756 60 +5757 51 +5758 51 +5759 65 +5760 76 +5761 65 +5762 67 +5763 73 +5764 61 +5765 78 +5766 62 +5767 59 +5768 63 +5769 63 +5770 62 +5771 77 +5772 57 +5773 69 +5774 58 +5775 60 +5776 52 +5777 48 +5778 57 +5779 55 +5780 63 +5781 52 +5782 69 +5783 59 +5784 56 +5785 57 +5786 64 +5787 65 +5788 60 +5789 64 +5790 54 +5791 75 +5792 52 +5793 52 +5794 65 +5795 67 +5796 59 +5797 61 +5798 81 +5799 52 +5800 59 +5801 53 +5802 61 +5803 56 +5804 63 +5805 58 +5806 69 +5807 52 +5808 54 +5809 60 +5810 49 +5811 56 +5812 52 +5813 67 +5814 68 +5815 80 +5816 60 +5817 69 +5818 56 +5819 52 +5820 56 +5821 60 +5822 59 +5823 71 +5824 58 +5825 52 +5826 59 +5827 59 +5828 72 +5829 61 +5830 45 +5831 60 +5832 68 +5833 62 +5834 57 +5835 60 +5836 66 +5837 72 +5838 58 +5839 71 +5840 55 +5841 55 +5842 55 +5843 53 +5844 66 +5845 44 +5846 54 +5847 55 +5848 44 +5849 65 +5850 58 +5851 52 +5852 53 +5853 56 +5854 69 +5855 53 +5856 57 +5857 59 +5858 52 +5859 60 +5860 52 +5861 65 +5862 58 +5863 65 +5864 60 +5865 67 +5866 51 +5867 60 +5868 54 +5869 56 +5870 50 +5871 68 +5872 55 +5873 57 +5874 52 +5875 54 +5876 54 +5877 55 +5878 44 +5879 74 +5880 48 +5881 63 +5882 60 +5883 51 +5884 69 +5885 58 +5886 53 +5887 61 +5888 54 +5889 56 +5890 74 +5891 61 +5892 57 +5893 52 +5894 53 +5895 50 +5896 63 +5897 53 +5898 55 +5899 57 +5900 59 +5901 53 +5902 53 +5903 57 +5904 59 +5905 64 +5906 49 +5907 66 +5908 57 +5909 56 +5910 61 +5911 54 +5912 66 +5913 58 +5914 68 +5915 63 +5916 63 +5917 65 +5918 38 +5919 49 +5920 74 +5921 59 +5922 56 +5923 50 +5924 56 +5925 52 +5926 64 +5927 58 +5928 52 +5929 54 +5930 56 +5931 57 +5932 56 +5933 67 +5934 57 +5935 44 +5936 62 +5937 49 +5938 67 +5939 50 +5940 54 +5941 58 +5942 51 +5943 69 +5944 62 +5945 51 +5946 62 +5947 68 +5948 53 +5949 69 +5950 62 +5951 70 +5952 62 +5953 57 +5954 77 +5955 55 +5956 45 +5957 47 +5958 55 +5959 60 +5960 52 +5961 51 +5962 54 +5963 58 +5964 43 +5965 66 +5966 60 +5967 52 +5968 56 +5969 64 +5970 67 +5971 52 +5972 57 +5973 57 +5974 59 +5975 52 +5976 51 +5977 45 +5978 57 +5979 64 +5980 64 +5981 60 +5982 69 +5983 62 +5984 67 +5985 53 +5986 63 +5987 47 +5988 62 +5989 49 +5990 52 +5991 68 +5992 57 +5993 56 +5994 43 +5995 53 +5996 49 +5997 63 +5998 62 +5999 47 +6000 68 +6001 56 +6002 60 +6003 55 +6004 52 +6005 59 +6006 59 +6007 47 +6008 65 +6009 54 +6010 59 +6011 70 +6012 54 +6013 63 +6014 67 +6015 55 +6016 57 +6017 49 +6018 55 +6019 51 +6020 63 +6021 65 +6022 50 +6023 54 +6024 51 +6025 36 +6026 55 +6027 47 +6028 62 +6029 64 +6030 45 +6031 45 +6032 48 +6033 51 +6034 57 +6035 65 +6036 54 +6037 47 +6038 50 +6039 62 +6040 55 +6041 58 +6042 46 +6043 73 +6044 56 +6045 46 +6046 62 +6047 65 +6048 50 +6049 64 +6050 43 +6051 60 +6052 61 +6053 53 +6054 57 +6055 50 +6056 66 +6057 64 +6058 43 +6059 49 +6060 55 +6061 49 +6062 56 +6063 57 +6064 46 +6065 57 +6066 46 +6067 39 +6068 47 +6069 50 +6070 65 +6071 56 +6072 52 +6073 63 +6074 63 +6075 58 +6076 60 +6077 58 +6078 32 +6079 46 +6080 50 +6081 49 +6082 54 +6083 51 +6084 40 +6085 54 +6086 46 +6087 42 +6088 57 +6089 49 +6090 55 +6091 56 +6092 60 +6093 50 +6094 46 +6095 55 +6096 55 +6097 53 +6098 50 +6099 50 +6100 47 +6101 65 +6102 50 +6103 49 +6104 58 +6105 67 +6106 69 +6107 48 +6108 49 +6109 51 +6110 53 +6111 45 +6112 55 +6113 60 +6114 47 +6115 59 +6116 58 +6117 40 +6118 49 +6119 38 +6120 47 +6121 67 +6122 53 +6123 52 +6124 51 +6125 55 +6126 65 +6127 52 +6128 61 +6129 54 +6130 63 +6131 38 +6132 56 +6133 71 +6134 46 +6135 52 +6136 52 +6137 59 +6138 50 +6139 54 +6140 41 +6141 53 +6142 61 +6143 64 +6144 49 +6145 45 +6146 35 +6147 76 +6148 72 +6149 53 +6150 64 +6151 51 +6152 51 +6153 59 +6154 58 +6155 57 +6156 66 +6157 49 +6158 48 +6159 55 +6160 41 +6161 57 +6162 46 +6163 48 +6164 52 +6165 52 +6166 60 +6167 51 +6168 68 +6169 67 +6170 49 +6171 53 +6172 61 +6173 62 +6174 58 +6175 57 +6176 48 +6177 54 +6178 53 +6179 63 +6180 45 +6181 64 +6182 60 +6183 47 +6184 57 +6185 55 +6186 63 +6187 60 +6188 59 +6189 55 +6190 53 +6191 48 +6192 56 +6193 49 +6194 58 +6195 54 +6196 50 +6197 64 +6198 54 +6199 46 +6200 61 +6201 60 +6202 56 +6203 56 +6204 57 +6205 60 +6206 45 +6207 65 +6208 56 +6209 57 +6210 45 +6211 48 +6212 48 +6213 58 +6214 38 +6215 61 +6216 51 +6217 56 +6218 55 +6219 51 +6220 56 +6221 48 +6222 63 +6223 50 +6224 53 +6225 62 +6226 55 +6227 46 +6228 47 +6229 41 +6230 52 +6231 65 +6232 81 +6233 57 +6234 55 +6235 59 +6236 46 +6237 53 +6238 75 +6239 66 +6240 40 +6241 51 +6242 67 +6243 51 +6244 57 +6245 39 +6246 61 +6247 56 +6248 55 +6249 56 +6250 38 +6251 55 +6252 59 +6253 56 +6254 48 +6255 48 +6256 44 +6257 61 +6258 43 +6259 47 +6260 55 +6261 52 +6262 64 +6263 61 +6264 54 +6265 45 +6266 46 +6267 52 +6268 49 +6269 61 +6270 50 +6271 59 +6272 43 +6273 61 +6274 52 +6275 30 +6276 51 +6277 56 +6278 56 +6279 53 +6280 52 +6281 65 +6282 53 +6283 58 +6284 54 +6285 56 +6286 60 +6287 48 +6288 74 +6289 58 +6290 52 +6291 50 +6292 57 +6293 56 +6294 60 +6295 57 +6296 51 +6297 39 +6298 56 +6299 50 +6300 49 +6301 39 +6302 58 +6303 33 +6304 51 +6305 56 +6306 48 +6307 50 +6308 67 +6309 50 +6310 55 +6311 57 +6312 57 +6313 42 +6314 56 +6315 60 +6316 59 +6317 57 +6318 43 +6319 49 +6320 50 +6321 53 +6322 54 +6323 42 +6324 62 +6325 56 +6326 55 +6327 41 +6328 42 +6329 55 +6330 48 +6331 51 +6332 55 +6333 53 +6334 39 +6335 44 +6336 63 +6337 48 +6338 57 +6339 50 +6340 53 +6341 60 +6342 52 +6343 51 +6344 65 +6345 50 +6346 62 +6347 49 +6348 52 +6349 69 +6350 49 +6351 60 +6352 52 +6353 50 +6354 59 +6355 60 +6356 51 +6357 56 +6358 51 +6359 39 +6360 49 +6361 53 +6362 57 +6363 46 +6364 57 +6365 52 +6366 53 +6367 38 +6368 51 +6369 51 +6370 48 +6371 53 +6372 50 +6373 52 +6374 49 +6375 56 +6376 40 +6377 52 +6378 60 +6379 52 +6380 53 +6381 55 +6382 47 +6383 59 +6384 48 +6385 45 +6386 57 +6387 53 +6388 56 +6389 54 +6390 50 +6391 58 +6392 50 +6393 58 +6394 50 +6395 51 +6396 53 +6397 45 +6398 44 +6399 56 +6400 65 +6401 34 +6402 44 +6403 54 +6404 45 +6405 55 +6406 57 +6407 54 +6408 57 +6409 54 +6410 49 +6411 60 +6412 41 +6413 45 +6414 58 +6415 49 +6416 47 +6417 37 +6418 48 +6419 51 +6420 53 +6421 54 +6422 50 +6423 64 +6424 51 +6425 54 +6426 55 +6427 63 +6428 62 +6429 46 +6430 51 +6431 44 +6432 55 +6433 40 +6434 49 +6435 59 +6436 56 +6437 49 +6438 46 +6439 61 +6440 53 +6441 48 +6442 65 +6443 53 +6444 58 +6445 55 +6446 57 +6447 59 +6448 65 +6449 44 +6450 42 +6451 57 +6452 48 +6453 48 +6454 47 +6455 60 +6456 59 +6457 44 +6458 47 +6459 39 +6460 42 +6461 52 +6462 55 +6463 63 +6464 60 +6465 54 +6466 60 +6467 56 +6468 48 +6469 49 +6470 60 +6471 45 +6472 47 +6473 45 +6474 50 +6475 50 +6476 47 +6477 57 +6478 50 +6479 51 +6480 58 +6481 44 +6482 51 +6483 59 +6484 52 +6485 62 +6486 45 +6487 59 +6488 46 +6489 42 +6490 52 +6491 53 +6492 47 +6493 38 +6494 47 +6495 50 +6496 55 +6497 52 +6498 39 +6499 41 +6500 45 +6501 55 +6502 48 +6503 37 +6504 49 +6505 44 +6506 62 +6507 37 +6508 62 +6509 47 +6510 38 +6511 46 +6512 62 +6513 54 +6514 51 +6515 47 +6516 44 +6517 38 +6518 51 +6519 54 +6520 44 +6521 57 +6522 48 +6523 43 +6524 48 +6525 35 +6526 49 +6527 59 +6528 56 +6529 41 +6530 44 +6531 50 +6532 56 +6533 55 +6534 49 +6535 46 +6536 45 +6537 53 +6538 46 +6539 39 +6540 59 +6541 52 +6542 57 +6543 45 +6544 51 +6545 46 +6546 40 +6547 55 +6548 43 +6549 51 +6550 52 +6551 46 +6552 48 +6553 51 +6554 44 +6555 39 +6556 49 +6557 48 +6558 61 +6559 42 +6560 67 +6561 46 +6562 64 +6563 42 +6564 40 +6565 56 +6566 44 +6567 48 +6568 50 +6569 53 +6570 51 +6571 39 +6572 45 +6573 39 +6574 50 +6575 44 +6576 44 +6577 50 +6578 50 +6579 51 +6580 48 +6581 48 +6582 45 +6583 61 +6584 57 +6585 56 +6586 38 +6587 46 +6588 39 +6589 48 +6590 46 +6591 40 +6592 43 +6593 63 +6594 49 +6595 43 +6596 59 +6597 39 +6598 50 +6599 52 +6600 55 +6601 48 +6602 54 +6603 54 +6604 61 +6605 42 +6606 46 +6607 53 +6608 40 +6609 44 +6610 47 +6611 38 +6612 59 +6613 42 +6614 36 +6615 47 +6616 50 +6617 48 +6618 53 +6619 43 +6620 34 +6621 58 +6622 53 +6623 50 +6624 32 +6625 41 +6626 49 +6627 33 +6628 42 +6629 38 +6630 36 +6631 52 +6632 42 +6633 44 +6634 34 +6635 52 +6636 55 +6637 54 +6638 38 +6639 48 +6640 49 +6641 43 +6642 46 +6643 52 +6644 54 +6645 39 +6646 46 +6647 41 +6648 43 +6649 50 +6650 44 +6651 47 +6652 41 +6653 40 +6654 43 +6655 48 +6656 36 +6657 42 +6658 38 +6659 40 +6660 46 +6661 48 +6662 30 +6663 45 +6664 48 +6665 42 +6666 49 +6667 39 +6668 45 +6669 49 +6670 43 +6671 43 +6672 45 +6673 57 +6674 53 +6675 59 +6676 49 +6677 30 +6678 42 +6679 48 +6680 48 +6681 41 +6682 40 +6683 48 +6684 59 +6685 38 +6686 49 +6687 45 +6688 56 +6689 48 +6690 47 +6691 48 +6692 45 +6693 43 +6694 52 +6695 50 +6696 38 +6697 46 +6698 40 +6699 39 +6700 49 +6701 39 +6702 41 +6703 51 +6704 31 +6705 55 +6706 51 +6707 47 +6708 59 +6709 49 +6710 43 +6711 52 +6712 44 +6713 49 +6714 40 +6715 53 +6716 51 +6717 43 +6718 39 +6719 43 +6720 49 +6721 41 +6722 44 +6723 43 +6724 31 +6725 50 +6726 58 +6727 45 +6728 40 +6729 55 +6730 38 +6731 45 +6732 53 +6733 55 +6734 47 +6735 35 +6736 38 +6737 41 +6738 52 +6739 41 +6740 33 +6741 45 +6742 45 +6743 44 +6744 53 +6745 41 +6746 48 +6747 47 +6748 41 +6749 39 +6750 56 +6751 42 +6752 47 +6753 50 +6754 52 +6755 49 +6756 43 +6757 47 +6758 45 +6759 54 +6760 45 +6761 51 +6762 50 +6763 40 +6764 49 +6765 56 +6766 41 +6767 52 +6768 60 +6769 52 +6770 32 +6771 54 +6772 33 +6773 54 +6774 40 +6775 56 +6776 50 +6777 40 +6778 45 +6779 33 +6780 42 +6781 42 +6782 42 +6783 53 +6784 42 +6785 43 +6786 42 +6787 42 +6788 45 +6789 43 +6790 42 +6791 44 +6792 43 +6793 47 +6794 44 +6795 52 +6796 38 +6797 36 +6798 45 +6799 39 +6800 66 +6801 47 +6802 53 +6803 39 +6804 59 +6805 40 +6806 51 +6807 42 +6808 27 +6809 55 +6810 61 +6811 52 +6812 46 +6813 51 +6814 50 +6815 43 +6816 43 +6817 42 +6818 45 +6819 39 +6820 48 +6821 46 +6822 50 +6823 39 +6824 46 +6825 64 +6826 53 +6827 39 +6828 36 +6829 38 +6830 46 +6831 45 +6832 40 +6833 51 +6834 43 +6835 40 +6836 40 +6837 51 +6838 44 +6839 48 +6840 31 +6841 45 +6842 30 +6843 37 +6844 39 +6845 52 +6846 47 +6847 43 +6848 56 +6849 42 +6850 46 +6851 50 +6852 41 +6853 31 +6854 47 +6855 51 +6856 41 +6857 34 +6858 45 +6859 45 +6860 39 +6861 44 +6862 46 +6863 49 +6864 42 +6865 49 +6866 54 +6867 34 +6868 41 +6869 44 +6870 38 +6871 43 +6872 38 +6873 47 +6874 45 +6875 37 +6876 37 +6877 46 +6878 39 +6879 43 +6880 43 +6881 39 +6882 43 +6883 40 +6884 54 +6885 36 +6886 40 +6887 45 +6888 35 +6889 45 +6890 31 +6891 40 +6892 47 +6893 44 +6894 42 +6895 47 +6896 34 +6897 45 +6898 44 +6899 42 +6900 44 +6901 42 +6902 31 +6903 47 +6904 42 +6905 51 +6906 37 +6907 46 +6908 48 +6909 47 +6910 48 +6911 50 +6912 45 +6913 39 +6914 48 +6915 55 +6916 37 +6917 44 +6918 47 +6919 45 +6920 47 +6921 50 +6922 40 +6923 41 +6924 49 +6925 45 +6926 57 +6927 58 +6928 47 +6929 37 +6930 41 +6931 43 +6932 45 +6933 55 +6934 34 +6935 40 +6936 45 +6937 42 +6938 40 +6939 49 +6940 43 +6941 33 +6942 42 +6943 42 +6944 44 +6945 40 +6946 37 +6947 41 +6948 44 +6949 53 +6950 42 +6951 45 +6952 38 +6953 46 +6954 39 +6955 43 +6956 49 +6957 43 +6958 45 +6959 44 +6960 44 +6961 45 +6962 42 +6963 34 +6964 43 +6965 49 +6966 48 +6967 58 +6968 54 +6969 43 +6970 47 +6971 54 +6972 46 +6973 52 +6974 45 +6975 56 +6976 30 +6977 56 +6978 55 +6979 44 +6980 46 +6981 36 +6982 37 +6983 42 +6984 48 +6985 35 +6986 41 +6987 48 +6988 46 +6989 34 +6990 41 +6991 50 +6992 50 +6993 41 +6994 39 +6995 44 +6996 40 +6997 41 +6998 33 +6999 48 +7000 35 +7001 40 +7002 51 +7003 35 +7004 49 +7005 39 +7006 41 +7007 43 +7008 34 +7009 43 +7010 47 +7011 31 +7012 39 +7013 47 +7014 38 +7015 42 +7016 32 +7017 35 +7018 36 +7019 38 +7020 32 +7021 33 +7022 42 +7023 39 +7024 40 +7025 44 +7026 42 +7027 35 +7028 54 +7029 52 +7030 42 +7031 53 +7032 46 +7033 46 +7034 43 +7035 43 +7036 42 +7037 26 +7038 40 +7039 49 +7040 49 +7041 54 +7042 42 +7043 50 +7044 43 +7045 58 +7046 44 +7047 35 +7048 44 +7049 48 +7050 38 +7051 43 +7052 28 +7053 50 +7054 30 +7055 46 +7056 41 +7057 43 +7058 43 +7059 32 +7060 36 +7061 28 +7062 46 +7063 46 +7064 55 +7065 37 +7066 45 +7067 26 +7068 50 +7069 42 +7070 47 +7071 51 +7072 45 +7073 31 +7074 42 +7075 47 +7076 44 +7077 40 +7078 45 +7079 31 +7080 43 +7081 34 +7082 33 +7083 37 +7084 47 +7085 32 +7086 50 +7087 47 +7088 54 +7089 31 +7090 42 +7091 31 +7092 38 +7093 49 +7094 42 +7095 51 +7096 45 +7097 38 +7098 38 +7099 45 +7100 51 +7101 46 +7102 46 +7103 49 +7104 41 +7105 32 +7106 63 +7107 38 +7108 38 +7109 40 +7110 34 +7111 34 +7112 44 +7113 40 +7114 34 +7115 55 +7116 58 +7117 48 +7118 41 +7119 40 +7120 30 +7121 35 +7122 34 +7123 48 +7124 38 +7125 42 +7126 35 +7127 38 +7128 40 +7129 42 +7130 27 +7131 38 +7132 35 +7133 40 +7134 50 +7135 55 +7136 40 +7137 41 +7138 42 +7139 40 +7140 40 +7141 35 +7142 49 +7143 41 +7144 41 +7145 33 +7146 44 +7147 46 +7148 47 +7149 42 +7150 50 +7151 39 +7152 59 +7153 48 +7154 43 +7155 42 +7156 38 +7157 43 +7158 49 +7159 48 +7160 34 +7161 43 +7162 42 +7163 44 +7164 45 +7165 36 +7166 48 +7167 38 +7168 48 +7169 43 +7170 41 +7171 50 +7172 28 +7173 34 +7174 37 +7175 36 +7176 33 +7177 42 +7178 27 +7179 60 +7180 55 +7181 45 +7182 32 +7183 37 +7184 38 +7185 47 +7186 45 +7187 40 +7188 35 +7189 51 +7190 52 +7191 42 +7192 39 +7193 36 +7194 54 +7195 45 +7196 41 +7197 38 +7198 36 +7199 41 +7200 51 +7201 51 +7202 47 +7203 45 +7204 41 +7205 42 +7206 37 +7207 42 +7208 46 +7209 36 +7210 40 +7211 44 +7212 36 +7213 37 +7214 50 +7215 30 +7216 42 +7217 42 +7218 44 +7219 35 +7220 45 +7221 38 +7222 32 +7223 46 +7224 55 +7225 49 +7226 50 +7227 46 +7228 35 +7229 34 +7230 45 +7231 48 +7232 52 +7233 36 +7234 39 +7235 38 +7236 38 +7237 44 +7238 36 +7239 29 +7240 37 +7241 48 +7242 35 +7243 37 +7244 43 +7245 31 +7246 31 +7247 34 +7248 46 +7249 39 +7250 37 +7251 43 +7252 48 +7253 38 +7254 42 +7255 47 +7256 45 +7257 43 +7258 51 +7259 42 +7260 49 +7261 41 +7262 57 +7263 49 +7264 44 +7265 34 +7266 39 +7267 34 +7268 39 +7269 53 +7270 34 +7271 40 +7272 42 +7273 38 +7274 53 +7275 40 +7276 46 +7277 46 +7278 41 +7279 51 +7280 41 +7281 35 +7282 51 +7283 48 +7284 41 +7285 47 +7286 56 +7287 41 +7288 42 +7289 33 +7290 45 +7291 35 +7292 34 +7293 49 +7294 32 +7295 30 +7296 43 +7297 36 +7298 38 +7299 36 +7300 37 +7301 40 +7302 45 +7303 46 +7304 40 +7305 38 +7306 41 +7307 45 +7308 45 +7309 37 +7310 32 +7311 35 +7312 38 +7313 43 +7314 38 +7315 35 +7316 44 +7317 34 +7318 44 +7319 42 +7320 37 +7321 46 +7322 40 +7323 43 +7324 35 +7325 41 +7326 44 +7327 37 +7328 36 +7329 50 +7330 36 +7331 42 +7332 40 +7333 31 +7334 44 +7335 37 +7336 37 +7337 44 +7338 37 +7339 31 +7340 41 +7341 40 +7342 43 +7343 48 +7344 48 +7345 41 +7346 44 +7347 39 +7348 49 +7349 37 +7350 46 +7351 46 +7352 37 +7353 42 +7354 51 +7355 45 +7356 42 +7357 45 +7358 43 +7359 57 +7360 43 +7361 40 +7362 51 +7363 36 +7364 38 +7365 39 +7366 44 +7367 42 +7368 37 +7369 54 +7370 47 +7371 40 +7372 39 +7373 37 +7374 47 +7375 38 +7376 31 +7377 33 +7378 44 +7379 29 +7380 30 +7381 34 +7382 39 +7383 33 +7384 42 +7385 29 +7386 41 +7387 45 +7388 38 +7389 46 +7390 49 +7391 38 +7392 38 +7393 50 +7394 46 +7395 42 +7396 35 +7397 42 +7398 31 +7399 47 +7400 44 +7401 44 +7402 44 +7403 46 +7404 43 +7405 37 +7406 33 +7407 39 +7408 27 +7409 46 +7410 43 +7411 53 +7412 30 +7413 44 +7414 51 +7415 45 +7416 35 +7417 30 +7418 31 +7419 43 +7420 25 +7421 33 +7422 38 +7423 37 +7424 43 +7425 33 +7426 29 +7427 43 +7428 38 +7429 26 +7430 49 +7431 40 +7432 41 +7433 41 +7434 44 +7435 30 +7436 32 +7437 36 +7438 41 +7439 37 +7440 43 +7441 54 +7442 33 +7443 39 +7444 40 +7445 46 +7446 51 +7447 39 +7448 32 +7449 34 +7450 36 +7451 50 +7452 31 +7453 22 +7454 48 +7455 37 +7456 35 +7457 36 +7458 38 +7459 38 +7460 40 +7461 40 +7462 39 +7463 43 +7464 35 +7465 46 +7466 33 +7467 32 +7468 42 +7469 33 +7470 34 +7471 43 +7472 39 +7473 32 +7474 30 +7475 33 +7476 48 +7477 29 +7478 35 +7479 43 +7480 45 +7481 48 +7482 38 +7483 28 +7484 30 +7485 31 +7486 42 +7487 35 +7488 38 +7489 40 +7490 37 +7491 45 +7492 57 +7493 43 +7494 31 +7495 46 +7496 39 +7497 44 +7498 31 +7499 32 +7500 36 +7501 37 +7502 33 +7503 46 +7504 29 +7505 38 +7506 41 +7507 44 +7508 31 +7509 46 +7510 36 +7511 32 +7512 35 +7513 32 +7514 36 +7515 43 +7516 34 +7517 42 +7518 46 +7519 41 +7520 35 +7521 35 +7522 51 +7523 44 +7524 45 +7525 44 +7526 45 +7527 39 +7528 33 +7529 33 +7530 36 +7531 51 +7532 39 +7533 36 +7534 25 +7535 52 +7536 40 +7537 44 +7538 44 +7539 49 +7540 39 +7541 39 +7542 30 +7543 33 +7544 46 +7545 37 +7546 36 +7547 38 +7548 43 +7549 44 +7550 38 +7551 35 +7552 42 +7553 39 +7554 48 +7555 32 +7556 34 +7557 49 +7558 38 +7559 35 +7560 27 +7561 45 +7562 37 +7563 32 +7564 35 +7565 36 +7566 28 +7567 49 +7568 27 +7569 44 +7570 48 +7571 39 +7572 40 +7573 52 +7574 33 +7575 48 +7576 42 +7577 45 +7578 42 +7579 46 +7580 29 +7581 27 +7582 33 +7583 30 +7584 42 +7585 42 +7586 41 +7587 35 +7588 48 +7589 32 +7590 47 +7591 38 +7592 37 +7593 31 +7594 33 +7595 46 +7596 43 +7597 38 +7598 43 +7599 37 +7600 30 +7601 42 +7602 32 +7603 46 +7604 30 +7605 43 +7606 37 +7607 31 +7608 30 +7609 47 +7610 37 +7611 31 +7612 35 +7613 33 +7614 30 +7615 30 +7616 30 +7617 23 +7618 42 +7619 47 +7620 34 +7621 32 +7622 33 +7623 38 +7624 30 +7625 31 +7626 42 +7627 47 +7628 47 +7629 27 +7630 34 +7631 45 +7632 37 +7633 48 +7634 36 +7635 47 +7636 42 +7637 35 +7638 31 +7639 42 +7640 36 +7641 35 +7642 42 +7643 33 +7644 43 +7645 36 +7646 34 +7647 39 +7648 37 +7649 29 +7650 35 +7651 44 +7652 35 +7653 39 +7654 27 +7655 29 +7656 34 +7657 33 +7658 27 +7659 38 +7660 39 +7661 46 +7662 36 +7663 26 +7664 38 +7665 37 +7666 38 +7667 39 +7668 29 +7669 49 +7670 43 +7671 30 +7672 41 +7673 42 +7674 35 +7675 37 +7676 32 +7677 39 +7678 36 +7679 37 +7680 31 +7681 35 +7682 33 +7683 40 +7684 32 +7685 26 +7686 41 +7687 40 +7688 42 +7689 41 +7690 37 +7691 33 +7692 42 +7693 35 +7694 31 +7695 27 +7696 32 +7697 33 +7698 35 +7699 40 +7700 39 +7701 28 +7702 47 +7703 37 +7704 39 +7705 41 +7706 48 +7707 32 +7708 48 +7709 33 +7710 30 +7711 33 +7712 34 +7713 37 +7714 26 +7715 42 +7716 36 +7717 34 +7718 39 +7719 26 +7720 37 +7721 41 +7722 31 +7723 29 +7724 37 +7725 26 +7726 28 +7727 37 +7728 25 +7729 37 +7730 32 +7731 43 +7732 45 +7733 34 +7734 26 +7735 38 +7736 27 +7737 30 +7738 34 +7739 36 +7740 41 +7741 51 +7742 29 +7743 36 +7744 26 +7745 32 +7746 31 +7747 32 +7748 30 +7749 28 +7750 35 +7751 29 +7752 39 +7753 31 +7754 23 +7755 28 +7756 39 +7757 34 +7758 32 +7759 24 +7760 40 +7761 41 +7762 29 +7763 34 +7764 35 +7765 37 +7766 33 +7767 26 +7768 37 +7769 34 +7770 24 +7771 29 +7772 50 +7773 24 +7774 36 +7775 29 +7776 45 +7777 43 +7778 31 +7779 37 +7780 35 +7781 42 +7782 41 +7783 33 +7784 41 +7785 34 +7786 47 +7787 36 +7788 38 +7789 24 +7790 31 +7791 33 +7792 34 +7793 41 +7794 22 +7795 40 +7796 49 +7797 31 +7798 42 +7799 29 +7800 29 +7801 29 +7802 45 +7803 32 +7804 37 +7805 29 +7806 42 +7807 36 +7808 26 +7809 36 +7810 28 +7811 40 +7812 41 +7813 36 +7814 36 +7815 34 +7816 40 +7817 28 +7818 40 +7819 27 +7820 28 +7821 40 +7822 32 +7823 42 +7824 31 +7825 37 +7826 40 +7827 30 +7828 38 +7829 29 +7830 38 +7831 35 +7832 32 +7833 26 +7834 29 +7835 37 +7836 29 +7837 26 +7838 32 +7839 32 +7840 37 +7841 29 +7842 30 +7843 34 +7844 33 +7845 33 +7846 49 +7847 32 +7848 31 +7849 35 +7850 37 +7851 29 +7852 44 +7853 25 +7854 39 +7855 38 +7856 25 +7857 31 +7858 41 +7859 25 +7860 26 +7861 34 +7862 28 +7863 27 +7864 34 +7865 27 +7866 35 +7867 30 +7868 35 +7869 31 +7870 37 +7871 20 +7872 32 +7873 25 +7874 27 +7875 43 +7876 27 +7877 36 +7878 42 +7879 26 +7880 30 +7881 26 +7882 43 +7883 31 +7884 37 +7885 33 +7886 35 +7887 37 +7888 35 +7889 40 +7890 31 +7891 34 +7892 32 +7893 36 +7894 34 +7895 31 +7896 37 +7897 42 +7898 34 +7899 33 +7900 37 +7901 47 +7902 28 +7903 29 +7904 29 +7905 36 +7906 39 +7907 34 +7908 34 +7909 35 +7910 28 +7911 44 +7912 40 +7913 28 +7914 25 +7915 35 +7916 28 +7917 31 +7918 31 +7919 34 +7920 39 +7921 42 +7922 41 +7923 37 +7924 40 +7925 31 +7926 26 +7927 35 +7928 30 +7929 32 +7930 18 +7931 36 +7932 41 +7933 32 +7934 39 +7935 34 +7936 30 +7937 30 +7938 34 +7939 36 +7940 33 +7941 40 +7942 37 +7943 25 +7944 27 +7945 37 +7946 24 +7947 17 +7948 36 +7949 38 +7950 28 +7951 38 +7952 27 +7953 35 +7954 32 +7955 25 +7956 32 +7957 31 +7958 32 +7959 40 +7960 34 +7961 33 +7962 26 +7963 43 +7964 39 +7965 37 +7966 37 +7967 36 +7968 29 +7969 39 +7970 32 +7971 34 +7972 31 +7973 36 +7974 34 +7975 34 +7976 24 +7977 32 +7978 28 +7979 30 +7980 36 +7981 36 +7982 43 +7983 31 +7984 37 +7985 40 +7986 34 +7987 29 +7988 37 +7989 31 +7990 35 +7991 23 +7992 32 +7993 34 +7994 38 +7995 25 +7996 24 +7997 32 +7998 31 +7999 31 +8000 37 +8001 34 +8002 29 +8003 31 +8004 35 +8005 32 +8006 31 +8007 39 +8008 32 +8009 33 +8010 37 +8011 31 +8012 26 +8013 33 +8014 36 +8015 32 +8016 39 +8017 31 +8018 29 +8019 44 +8020 37 +8021 42 +8022 25 +8023 29 +8024 26 +8025 34 +8026 25 +8027 30 +8028 31 +8029 34 +8030 24 +8031 37 +8032 37 +8033 35 +8034 27 +8035 31 +8036 33 +8037 34 +8038 33 +8039 28 +8040 31 +8041 33 +8042 30 +8043 32 +8044 30 +8045 36 +8046 41 +8047 40 +8048 34 +8049 30 +8050 29 +8051 33 +8052 26 +8053 34 +8054 32 +8055 33 +8056 27 +8057 27 +8058 17 +8059 29 +8060 30 +8061 31 +8062 43 +8063 27 +8064 31 +8065 27 +8066 28 +8067 32 +8068 24 +8069 20 +8070 24 +8071 25 +8072 24 +8073 26 +8074 39 +8075 30 +8076 25 +8077 43 +8078 39 +8079 33 +8080 32 +8081 32 +8082 33 +8083 24 +8084 36 +8085 37 +8086 35 +8087 38 +8088 24 +8089 32 +8090 41 +8091 24 +8092 36 +8093 45 +8094 30 +8095 35 +8096 29 +8097 44 +8098 21 +8099 35 +8100 28 +8101 27 +8102 21 +8103 24 +8104 41 +8105 36 +8106 33 +8107 32 +8108 30 +8109 32 +8110 23 +8111 32 +8112 24 +8113 35 +8114 33 +8115 36 +8116 29 +8117 37 +8118 33 +8119 31 +8120 36 +8121 35 +8122 33 +8123 31 +8124 35 +8125 21 +8126 41 +8127 28 +8128 36 +8129 35 +8130 29 +8131 27 +8132 39 +8133 30 +8134 26 +8135 26 +8136 27 +8137 29 +8138 24 +8139 38 +8140 26 +8141 43 +8142 42 +8143 22 +8144 39 +8145 24 +8146 24 +8147 31 +8148 37 +8149 33 +8150 29 +8151 27 +8152 36 +8153 31 +8154 30 +8155 31 +8156 24 +8157 29 +8158 25 +8159 39 +8160 21 +8161 29 +8162 22 +8163 30 +8164 37 +8165 30 +8166 24 +8167 26 +8168 30 +8169 28 +8170 32 +8171 33 +8172 39 +8173 35 +8174 39 +8175 35 +8176 41 +8177 24 +8178 41 +8179 36 +8180 38 +8181 25 +8182 23 +8183 32 +8184 32 +8185 27 +8186 33 +8187 27 +8188 31 +8189 27 +8190 30 +8191 32 +8192 36 +8193 41 +8194 35 +8195 23 +8196 33 +8197 27 +8198 26 +8199 32 +8200 41 +8201 33 +8202 32 +8203 23 +8204 27 +8205 27 +8206 34 +8207 38 +8208 19 +8209 40 +8210 31 +8211 29 +8212 27 +8213 28 +8214 32 +8215 36 +8216 25 +8217 27 +8218 36 +8219 37 +8220 22 +8221 29 +8222 30 +8223 21 +8224 19 +8225 25 +8226 26 +8227 27 +8228 28 +8229 23 +8230 20 +8231 30 +8232 38 +8233 35 +8234 38 +8235 28 +8236 27 +8237 27 +8238 29 +8239 30 +8240 35 +8241 31 +8242 25 +8243 29 +8244 27 +8245 40 +8246 30 +8247 37 +8248 32 +8249 32 +8250 31 +8251 24 +8252 27 +8253 35 +8254 34 +8255 29 +8256 28 +8257 32 +8258 27 +8259 26 +8260 20 +8261 27 +8262 29 +8263 32 +8264 31 +8265 19 +8266 26 +8267 29 +8268 29 +8269 30 +8270 31 +8271 23 +8272 32 +8273 27 +8274 35 +8275 18 +8276 24 +8277 26 +8278 29 +8279 24 +8280 31 +8281 34 +8282 31 +8283 24 +8284 31 +8285 30 +8286 27 +8287 25 +8288 37 +8289 36 +8290 27 +8291 33 +8292 28 +8293 24 +8294 24 +8295 26 +8296 30 +8297 41 +8298 24 +8299 22 +8300 30 +8301 36 +8302 39 +8303 23 +8304 31 +8305 24 +8306 26 +8307 27 +8308 34 +8309 36 +8310 25 +8311 33 +8312 25 +8313 31 +8314 28 +8315 26 +8316 23 +8317 26 +8318 30 +8319 29 +8320 26 +8321 30 +8322 32 +8323 29 +8324 17 +8325 21 +8326 27 +8327 38 +8328 28 +8329 26 +8330 36 +8331 36 +8332 29 +8333 25 +8334 23 +8335 27 +8336 38 +8337 28 +8338 28 +8339 30 +8340 23 +8341 32 +8342 29 +8343 30 +8344 20 +8345 32 +8346 37 +8347 29 +8348 29 +8349 29 +8350 32 +8351 27 +8352 22 +8353 23 +8354 30 +8355 33 +8356 30 +8357 33 +8358 16 +8359 23 +8360 25 +8361 36 +8362 27 +8363 36 +8364 30 +8365 28 +8366 33 +8367 32 +8368 24 +8369 24 +8370 40 +8371 28 +8372 32 +8373 29 +8374 35 +8375 34 +8376 25 +8377 30 +8378 27 +8379 35 +8380 27 +8381 22 +8382 21 +8383 24 +8384 36 +8385 31 +8386 31 +8387 26 +8388 25 +8389 29 +8390 30 +8391 31 +8392 22 +8393 25 +8394 25 +8395 34 +8396 29 +8397 34 +8398 33 +8399 34 +8400 35 +8401 24 +8402 23 +8403 26 +8404 29 +8405 29 +8406 30 +8407 26 +8408 41 +8409 34 +8410 29 +8411 28 +8412 28 +8413 31 +8414 33 +8415 34 +8416 34 +8417 22 +8418 27 +8419 34 +8420 18 +8421 31 +8422 35 +8423 34 +8424 29 +8425 26 +8426 31 +8427 34 +8428 22 +8429 17 +8430 34 +8431 29 +8432 35 +8433 36 +8434 27 +8435 31 +8436 27 +8437 27 +8438 33 +8439 29 +8440 31 +8441 27 +8442 30 +8443 22 +8444 27 +8445 37 +8446 24 +8447 31 +8448 30 +8449 37 +8450 37 +8451 20 +8452 25 +8453 31 +8454 23 +8455 21 +8456 30 +8457 34 +8458 30 +8459 34 +8460 32 +8461 32 +8462 24 +8463 24 +8464 34 +8465 22 +8466 26 +8467 29 +8468 22 +8469 29 +8470 32 +8471 19 +8472 30 +8473 28 +8474 32 +8475 18 +8476 31 +8477 31 +8478 34 +8479 32 +8480 34 +8481 23 +8482 28 +8483 38 +8484 31 +8485 22 +8486 30 +8487 33 +8488 26 +8489 25 +8490 26 +8491 34 +8492 32 +8493 28 +8494 27 +8495 29 +8496 21 +8497 29 +8498 30 +8499 30 +8500 30 +8501 25 +8502 31 +8503 43 +8504 24 +8505 25 +8506 33 +8507 33 +8508 24 +8509 34 +8510 26 +8511 15 +8512 29 +8513 22 +8514 32 +8515 34 +8516 31 +8517 34 +8518 33 +8519 27 +8520 31 +8521 24 +8522 20 +8523 32 +8524 34 +8525 32 +8526 27 +8527 29 +8528 22 +8529 31 +8530 23 +8531 31 +8532 29 +8533 25 +8534 23 +8535 29 +8536 24 +8537 37 +8538 33 +8539 30 +8540 31 +8541 27 +8542 19 +8543 35 +8544 35 +8545 26 +8546 25 +8547 26 +8548 29 +8549 25 +8550 28 +8551 30 +8552 34 +8553 28 +8554 21 +8555 36 +8556 21 +8557 27 +8558 17 +8559 29 +8560 24 +8561 33 +8562 22 +8563 23 +8564 17 +8565 29 +8566 40 +8567 32 +8568 17 +8569 24 +8570 24 +8571 29 +8572 23 +8573 31 +8574 24 +8575 24 +8576 22 +8577 25 +8578 18 +8579 25 +8580 35 +8581 42 +8582 23 +8583 23 +8584 25 +8585 26 +8586 34 +8587 35 +8588 24 +8589 26 +8590 23 +8591 25 +8592 29 +8593 37 +8594 25 +8595 22 +8596 30 +8597 30 +8598 23 +8599 35 +8600 28 +8601 22 +8602 26 +8603 29 +8604 21 +8605 29 +8606 26 +8607 28 +8608 32 +8609 33 +8610 30 +8611 39 +8612 35 +8613 32 +8614 29 +8615 37 +8616 27 +8617 31 +8618 29 +8619 20 +8620 24 +8621 24 +8622 26 +8623 26 +8624 34 +8625 31 +8626 32 +8627 22 +8628 33 +8629 29 +8630 34 +8631 27 +8632 24 +8633 22 +8634 25 +8635 37 +8636 31 +8637 32 +8638 34 +8639 32 +8640 37 +8641 23 +8642 38 +8643 23 +8644 32 +8645 30 +8646 27 +8647 33 +8648 26 +8649 16 +8650 34 +8651 22 +8652 31 +8653 15 +8654 20 +8655 19 +8656 27 +8657 31 +8658 20 +8659 33 +8660 28 +8661 34 +8662 23 +8663 30 +8664 28 +8665 27 +8666 42 +8667 32 +8668 23 +8669 25 +8670 20 +8671 26 +8672 41 +8673 29 +8674 30 +8675 30 +8676 27 +8677 28 +8678 32 +8679 37 +8680 36 +8681 30 +8682 36 +8683 38 +8684 36 +8685 30 +8686 18 +8687 24 +8688 30 +8689 32 +8690 27 +8691 26 +8692 34 +8693 24 +8694 16 +8695 21 +8696 24 +8697 23 +8698 23 +8699 34 +8700 23 +8701 24 +8702 24 +8703 23 +8704 35 +8705 33 +8706 33 +8707 19 +8708 31 +8709 30 +8710 25 +8711 28 +8712 18 +8713 38 +8714 28 +8715 41 +8716 31 +8717 36 +8718 28 +8719 30 +8720 36 +8721 33 +8722 23 +8723 21 +8724 29 +8725 30 +8726 26 +8727 27 +8728 31 +8729 19 +8730 27 +8731 33 +8732 26 +8733 28 +8734 27 +8735 31 +8736 29 +8737 31 +8738 32 +8739 25 +8740 29 +8741 23 +8742 27 +8743 22 +8744 32 +8745 30 +8746 29 +8747 33 +8748 33 +8749 33 +8750 25 +8751 31 +8752 31 +8753 36 +8754 33 +8755 34 +8756 26 +8757 33 +8758 32 +8759 31 +8760 31 +8761 29 +8762 30 +8763 34 +8764 21 +8765 17 +8766 31 +8767 26 +8768 22 +8769 28 +8770 22 +8771 32 +8772 28 +8773 21 +8774 26 +8775 16 +8776 26 +8777 32 +8778 28 +8779 19 +8780 21 +8781 27 +8782 25 +8783 25 +8784 31 +8785 26 +8786 30 +8787 22 +8788 21 +8789 27 +8790 39 +8791 25 +8792 28 +8793 24 +8794 28 +8795 23 +8796 31 +8797 34 +8798 30 +8799 33 +8800 30 +8801 23 +8802 24 +8803 37 +8804 30 +8805 31 +8806 32 +8807 25 +8808 18 +8809 27 +8810 30 +8811 26 +8812 27 +8813 30 +8814 39 +8815 27 +8816 38 +8817 24 +8818 27 +8819 37 +8820 23 +8821 21 +8822 26 +8823 25 +8824 19 +8825 22 +8826 29 +8827 24 +8828 26 +8829 35 +8830 36 +8831 27 +8832 32 +8833 17 +8834 19 +8835 29 +8836 29 +8837 29 +8838 32 +8839 25 +8840 30 +8841 29 +8842 27 +8843 32 +8844 26 +8845 24 +8846 31 +8847 31 +8848 23 +8849 17 +8850 29 +8851 37 +8852 31 +8853 20 +8854 31 +8855 28 +8856 21 +8857 29 +8858 29 +8859 23 +8860 21 +8861 33 +8862 21 +8863 33 +8864 21 +8865 18 +8866 20 +8867 31 +8868 28 +8869 31 +8870 25 +8871 32 +8872 36 +8873 28 +8874 30 +8875 26 +8876 18 +8877 16 +8878 27 +8879 30 +8880 19 +8881 26 +8882 28 +8883 27 +8884 28 +8885 28 +8886 37 +8887 27 +8888 23 +8889 21 +8890 17 +8891 27 +8892 28 +8893 31 +8894 30 +8895 28 +8896 27 +8897 28 +8898 38 +8899 22 +8900 29 +8901 30 +8902 18 +8903 20 +8904 33 +8905 31 +8906 24 +8907 24 +8908 39 +8909 20 +8910 16 +8911 22 +8912 28 +8913 29 +8914 28 +8915 29 +8916 28 +8917 26 +8918 26 +8919 27 +8920 33 +8921 28 +8922 27 +8923 23 +8924 27 +8925 20 +8926 23 +8927 19 +8928 17 +8929 31 +8930 25 +8931 37 +8932 17 +8933 27 +8934 28 +8935 24 +8936 24 +8937 23 +8938 20 +8939 18 +8940 13 +8941 24 +8942 30 +8943 25 +8944 32 +8945 29 +8946 27 +8947 32 +8948 23 +8949 22 +8950 31 +8951 20 +8952 22 +8953 27 +8954 28 +8955 24 +8956 25 +8957 19 +8958 23 +8959 17 +8960 33 +8961 21 +8962 26 +8963 27 +8964 27 +8965 13 +8966 28 +8967 24 +8968 40 +8969 33 +8970 21 +8971 25 +8972 27 +8973 19 +8974 19 +8975 33 +8976 23 +8977 23 +8978 21 +8979 26 +8980 24 +8981 22 +8982 37 +8983 27 +8984 28 +8985 17 +8986 24 +8987 28 +8988 24 +8989 26 +8990 26 +8991 25 +8992 25 +8993 14 +8994 31 +8995 14 +8996 24 +8997 27 +8998 20 +8999 29 +9000 25 +9001 30 +9002 25 +9003 26 +9004 28 +9005 25 +9006 30 +9007 25 +9008 29 +9009 26 +9010 24 +9011 31 +9012 31 +9013 28 +9014 28 +9015 31 +9016 18 +9017 26 +9018 29 +9019 19 +9020 29 +9021 21 +9022 21 +9023 32 +9024 38 +9025 25 +9026 35 +9027 18 +9028 18 +9029 30 +9030 30 +9031 29 +9032 21 +9033 34 +9034 19 +9035 32 +9036 27 +9037 34 +9038 23 +9039 23 +9040 20 +9041 30 +9042 23 +9043 27 +9044 22 +9045 32 +9046 24 +9047 21 +9048 27 +9049 25 +9050 19 +9051 28 +9052 19 +9053 28 +9054 33 +9055 17 +9056 22 +9057 29 +9058 21 +9059 32 +9060 30 +9061 16 +9062 33 +9063 24 +9064 33 +9065 27 +9066 22 +9067 29 +9068 27 +9069 23 +9070 21 +9071 25 +9072 22 +9073 28 +9074 29 +9075 28 +9076 30 +9077 24 +9078 30 +9079 18 +9080 26 +9081 29 +9082 21 +9083 32 +9084 29 +9085 20 +9086 22 +9087 21 +9088 28 +9089 28 +9090 24 +9091 26 +9092 20 +9093 27 +9094 28 +9095 21 +9096 29 +9097 19 +9098 26 +9099 26 +9100 31 +9101 30 +9102 25 +9103 31 +9104 31 +9105 20 +9106 26 +9107 25 +9108 22 +9109 19 +9110 29 +9111 23 +9112 24 +9113 28 +9114 29 +9115 27 +9116 25 +9117 22 +9118 30 +9119 20 +9120 29 +9121 28 +9122 26 +9123 32 +9124 37 +9125 32 +9126 30 +9127 32 +9128 24 +9129 27 +9130 17 +9131 17 +9132 33 +9133 24 +9134 28 +9135 23 +9136 21 +9137 25 +9138 19 +9139 32 +9140 19 +9141 23 +9142 15 +9143 25 +9144 32 +9145 24 +9146 27 +9147 29 +9148 24 +9149 29 +9150 31 +9151 25 +9152 28 +9153 15 +9154 28 +9155 22 +9156 23 +9157 25 +9158 23 +9159 26 +9160 24 +9161 31 +9162 22 +9163 35 +9164 27 +9165 37 +9166 26 +9167 21 +9168 23 +9169 21 +9170 25 +9171 28 +9172 28 +9173 23 +9174 28 +9175 22 +9176 27 +9177 22 +9178 24 +9179 31 +9180 25 +9181 30 +9182 26 +9183 33 +9184 24 +9185 25 +9186 23 +9187 25 +9188 19 +9189 28 +9190 34 +9191 32 +9192 18 +9193 33 +9194 25 +9195 25 +9196 27 +9197 23 +9198 26 +9199 18 +9200 18 +9201 22 +9202 21 +9203 23 +9204 22 +9205 23 +9206 23 +9207 24 +9208 29 +9209 22 +9210 21 +9211 30 +9212 23 +9213 16 +9214 25 +9215 21 +9216 32 +9217 21 +9218 21 +9219 27 +9220 30 +9221 27 +9222 16 +9223 27 +9224 26 +9225 30 +9226 29 +9227 30 +9228 22 +9229 29 +9230 18 +9231 24 +9232 24 +9233 20 +9234 21 +9235 19 +9236 26 +9237 22 +9238 22 +9239 34 +9240 27 +9241 24 +9242 20 +9243 24 +9244 31 +9245 26 +9246 20 +9247 31 +9248 18 +9249 22 +9250 32 +9251 21 +9252 20 +9253 27 +9254 27 +9255 28 +9256 16 +9257 32 +9258 42 +9259 19 +9260 19 +9261 32 +9262 27 +9263 32 +9264 30 +9265 18 +9266 22 +9267 24 +9268 26 +9269 15 +9270 17 +9271 31 +9272 24 +9273 24 +9274 22 +9275 14 +9276 21 +9277 15 +9278 28 +9279 23 +9280 19 +9281 27 +9282 23 +9283 21 +9284 20 +9285 19 +9286 24 +9287 26 +9288 17 +9289 24 +9290 34 +9291 31 +9292 27 +9293 21 +9294 28 +9295 23 +9296 25 +9297 24 +9298 24 +9299 32 +9300 28 +9301 29 +9302 30 +9303 24 +9304 25 +9305 23 +9306 14 +9307 31 +9308 28 +9309 17 +9310 23 +9311 30 +9312 28 +9313 27 +9314 20 +9315 21 +9316 18 +9317 14 +9318 24 +9319 27 +9320 28 +9321 27 +9322 22 +9323 20 +9324 26 +9325 32 +9326 28 +9327 27 +9328 21 +9329 24 +9330 18 +9331 23 +9332 24 +9333 29 +9334 27 +9335 20 +9336 29 +9337 27 +9338 27 +9339 22 +9340 20 +9341 19 +9342 22 +9343 26 +9344 14 +9345 24 +9346 20 +9347 30 +9348 24 +9349 28 +9350 22 +9351 18 +9352 15 +9353 30 +9354 25 +9355 26 +9356 21 +9357 19 +9358 23 +9359 18 +9360 19 +9361 21 +9362 26 +9363 22 +9364 36 +9365 27 +9366 24 +9367 22 +9368 31 +9369 19 +9370 24 +9371 20 +9372 28 +9373 23 +9374 25 +9375 24 +9376 23 +9377 22 +9378 22 +9379 20 +9380 18 +9381 26 +9382 15 +9383 25 +9384 30 +9385 20 +9386 26 +9387 22 +9388 20 +9389 22 +9390 16 +9391 15 +9392 34 +9393 12 +9394 23 +9395 21 +9396 19 +9397 22 +9398 17 +9399 28 +9400 20 +9401 21 +9402 24 +9403 28 +9404 22 +9405 24 +9406 16 +9407 29 +9408 19 +9409 19 +9410 15 +9411 26 +9412 25 +9413 26 +9414 26 +9415 23 +9416 25 +9417 27 +9418 26 +9419 13 +9420 20 +9421 35 +9422 19 +9423 21 +9424 22 +9425 18 +9426 22 +9427 19 +9428 23 +9429 24 +9430 22 +9431 27 +9432 22 +9433 19 +9434 31 +9435 22 +9436 12 +9437 23 +9438 19 +9439 25 +9440 20 +9441 18 +9442 21 +9443 25 +9444 35 +9445 20 +9446 20 +9447 21 +9448 23 +9449 21 +9450 32 +9451 23 +9452 16 +9453 21 +9454 30 +9455 25 +9456 18 +9457 23 +9458 19 +9459 30 +9460 17 +9461 18 +9462 23 +9463 20 +9464 20 +9465 23 +9466 30 +9467 19 +9468 18 +9469 23 +9470 26 +9471 21 +9472 13 +9473 23 +9474 18 +9475 14 +9476 25 +9477 25 +9478 18 +9479 14 +9480 22 +9481 20 +9482 26 +9483 21 +9484 24 +9485 22 +9486 25 +9487 27 +9488 21 +9489 13 +9490 27 +9491 13 +9492 25 +9493 15 +9494 23 +9495 22 +9496 24 +9497 23 +9498 22 +9499 18 +9500 24 +9501 24 +9502 22 +9503 27 +9504 23 +9505 18 +9506 34 +9507 17 +9508 19 +9509 25 +9510 24 +9511 27 +9512 19 +9513 20 +9514 21 +9515 18 +9516 26 +9517 11 +9518 21 +9519 29 +9520 21 +9521 24 +9522 30 +9523 24 +9524 21 +9525 30 +9526 22 +9527 28 +9528 26 +9529 22 +9530 26 +9531 19 +9532 26 +9533 11 +9534 26 +9535 24 +9536 18 +9537 14 +9538 25 +9539 28 +9540 28 +9541 20 +9542 31 +9543 14 +9544 24 +9545 21 +9546 23 +9547 23 +9548 22 +9549 15 +9550 24 +9551 22 +9552 27 +9553 23 +9554 19 +9555 22 +9556 24 +9557 16 +9558 23 +9559 18 +9560 18 +9561 29 +9562 22 +9563 25 +9564 16 +9565 29 +9566 30 +9567 21 +9568 31 +9569 32 +9570 16 +9571 19 +9572 15 +9573 29 +9574 31 +9575 12 +9576 26 +9577 30 +9578 20 +9579 18 +9580 26 +9581 33 +9582 30 +9583 23 +9584 21 +9585 28 +9586 19 +9587 21 +9588 20 +9589 23 +9590 33 +9591 21 +9592 24 +9593 26 +9594 23 +9595 17 +9596 27 +9597 24 +9598 27 +9599 19 +9600 24 +9601 19 +9602 18 +9603 28 +9604 26 +9605 27 +9606 20 +9607 18 +9608 16 +9609 17 +9610 21 +9611 28 +9612 31 +9613 17 +9614 27 +9615 30 +9616 15 +9617 26 +9618 26 +9619 24 +9620 29 +9621 14 +9622 23 +9623 22 +9624 25 +9625 27 +9626 21 +9627 21 +9628 23 +9629 20 +9630 26 +9631 24 +9632 17 +9633 30 +9634 20 +9635 30 +9636 19 +9637 19 +9638 23 +9639 16 +9640 27 +9641 21 +9642 20 +9643 28 +9644 30 +9645 16 +9646 25 +9647 33 +9648 21 +9649 20 +9650 15 +9651 18 +9652 14 +9653 25 +9654 26 +9655 27 +9656 21 +9657 18 +9658 14 +9659 18 +9660 19 +9661 22 +9662 25 +9663 17 +9664 21 +9665 22 +9666 21 +9667 23 +9668 16 +9669 16 +9670 26 +9671 14 +9672 23 +9673 40 +9674 20 +9675 26 +9676 30 +9677 30 +9678 24 +9679 21 +9680 26 +9681 28 +9682 18 +9683 13 +9684 23 +9685 22 +9686 28 +9687 17 +9688 17 +9689 24 +9690 23 +9691 16 +9692 21 +9693 20 +9694 18 +9695 16 +9696 20 +9697 21 +9698 32 +9699 23 +9700 12 +9701 17 +9702 17 +9703 17 +9704 15 +9705 18 +9706 20 +9707 25 +9708 26 +9709 18 +9710 30 +9711 17 +9712 20 +9713 28 +9714 21 +9715 27 +9716 22 +9717 20 +9718 18 +9719 23 +9720 28 +9721 21 +9722 25 +9723 19 +9724 20 +9725 18 +9726 23 +9727 23 +9728 26 +9729 22 +9730 23 +9731 18 +9732 17 +9733 28 +9734 23 +9735 18 +9736 19 +9737 20 +9738 29 +9739 21 +9740 26 +9741 33 +9742 17 +9743 21 +9744 23 +9745 32 +9746 32 +9747 20 +9748 30 +9749 28 +9750 23 +9751 20 +9752 21 +9753 16 +9754 22 +9755 16 +9756 20 +9757 23 +9758 19 +9759 24 +9760 13 +9761 18 +9762 16 +9763 16 +9764 17 +9765 25 +9766 15 +9767 20 +9768 15 +9769 21 +9770 23 +9771 21 +9772 24 +9773 23 +9774 27 +9775 28 +9776 17 +9777 26 +9778 17 +9779 22 +9780 18 +9781 28 +9782 15 +9783 21 +9784 15 +9785 21 +9786 21 +9787 21 +9788 24 +9789 22 +9790 24 +9791 16 +9792 22 +9793 14 +9794 21 +9795 27 +9796 26 +9797 26 +9798 25 +9799 27 +9800 18 +9801 16 +9802 15 +9803 19 +9804 19 +9805 26 +9806 20 +9807 17 +9808 24 +9809 19 +9810 16 +9811 25 +9812 31 +9813 25 +9814 24 +9815 21 +9816 22 +9817 18 +9818 21 +9819 12 +9820 18 +9821 30 +9822 22 +9823 29 +9824 20 +9825 23 +9826 32 +9827 29 +9828 19 +9829 23 +9830 24 +9831 18 +9832 24 +9833 18 +9834 24 +9835 19 +9836 16 +9837 23 +9838 24 +9839 25 +9840 19 +9841 20 +9842 17 +9843 15 +9844 23 +9845 16 +9846 24 +9847 18 +9848 35 +9849 27 +9850 23 +9851 22 +9852 24 +9853 11 +9854 18 +9855 21 +9856 23 +9857 16 +9858 25 +9859 19 +9860 18 +9861 22 +9862 28 +9863 20 +9864 20 +9865 25 +9866 17 +9867 20 +9868 24 +9869 25 +9870 22 +9871 23 +9872 19 +9873 22 +9874 23 +9875 21 +9876 26 +9877 17 +9878 26 +9879 15 +9880 25 +9881 22 +9882 21 +9883 20 +9884 14 +9885 16 +9886 27 +9887 33 +9888 23 +9889 21 +9890 22 +9891 30 +9892 18 +9893 26 +9894 22 +9895 16 +9896 14 +9897 19 +9898 25 +9899 24 +9900 20 +9901 28 +9902 21 +9903 25 +9904 26 +9905 21 +9906 19 +9907 19 +9908 23 +9909 21 +9910 17 +9911 26 +9912 28 +9913 16 +9914 23 +9915 22 +9916 25 +9917 26 +9918 13 +9919 9 +9920 23 +9921 25 +9922 20 +9923 18 +9924 23 +9925 18 +9926 22 +9927 26 +9928 21 +9929 31 +9930 25 +9931 15 +9932 24 +9933 13 +9934 19 +9935 15 +9936 17 +9937 24 +9938 24 +9939 22 +9940 30 +9941 18 +9942 19 +9943 22 +9944 18 +9945 21 +9946 20 +9947 19 +9948 28 +9949 24 +9950 19 +9951 16 +9952 20 +9953 18 +9954 16 +9955 23 +9956 25 +9957 24 +9958 23 +9959 12 +9960 25 +9961 15 +9962 24 +9963 25 +9964 21 +9965 25 +9966 28 +9967 22 +9968 14 +9969 22 +9970 22 +9971 24 +9972 20 +9973 24 +9974 18 +9975 18 +9976 20 +9977 9 +9978 15 +9979 18 +9980 21 +9981 12 +9982 25 +9983 21 +9984 17 +9985 15 +9986 21 +9987 28 +9988 19 +9989 25 +9990 29 +9991 19 +9992 24 +9993 22 +9994 16 +9995 21 +9996 13 +9997 21 +9998 20 +9999 23 +10000 23 +10001 172133 diff --git a/99.scripts/trinity_utils/util/misc/insilico_norm_kmer_hists/plot_me.R b/99.scripts/trinity_utils/util/misc/insilico_norm_kmer_hists/plot_me.R new file mode 100644 index 0000000..e120939 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/insilico_norm_kmer_hists/plot_me.R @@ -0,0 +1,9 @@ +data_all = read.table("kmer_histo.all.txt") +data_norm = read.table("kmer_histo.NormMaxKCov50.txt") + +log_data_all = cbind(data_all[,1], log(data_all[,2]+1)) +log_data_norm = cbind(data_norm[,1], log(data_norm[,2]+1)) + +plot(log_data_norm, col='green', xlim=c(0,200), xlab="kmer occurrence count", ylab="log(number of unique kmers)", main="Kmer composition and \nin silico read normalization") +points(log_data_all, col='red') + diff --git a/99.scripts/trinity_utils/util/misc/insilico_norm_kmer_hists/result.pdf b/99.scripts/trinity_utils/util/misc/insilico_norm_kmer_hists/result.pdf new file mode 100644 index 0000000..1809aa9 Binary files /dev/null and b/99.scripts/trinity_utils/util/misc/insilico_norm_kmer_hists/result.pdf differ diff --git a/99.scripts/trinity_utils/util/misc/iso_reco_analysis/bam_to_cuff.pl b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/bam_to_cuff.pl new file mode 100644 index 0000000..ded00e3 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/bam_to_cuff.pl @@ -0,0 +1,25 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use File::Basename; + +my $usage = "usage: $0 bams.list\n\n"; + +my $bam_list_file = $ARGV[0] or die $usage; + +open (my $fh, $bam_list_file) or die $!; +while (<$fh>) { + chomp; + my $bam_file = $_; + + my $base_dir = dirname($bam_file); + + my $cmd = "cufflinks -o $base_dir/ $bam_file"; + + print "$cmd\n"; +} + +exit(0); + + diff --git a/99.scripts/trinity_utils/util/misc/iso_reco_analysis/cuff_gtf_to_bed.pl b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/cuff_gtf_to_bed.pl new file mode 100644 index 0000000..ac58a1d --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/cuff_gtf_to_bed.pl @@ -0,0 +1,28 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 cuff_gtf.list\n\n"; + +my $trans_gtf_files = $ARGV[0] or die $usage; + +main: { + + open (my $fh, $trans_gtf_files) or die $!; + + while (<$fh>) { + chomp; + my $filename = $_; + + my $cmd = "cufflinks_gtf_to_bed.pl $filename > $filename.bed"; + + print "$cmd\n"; + } + + close $fh; + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/iso_reco_analysis/gene_gff3_to_bed_cmds.pl b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/gene_gff3_to_bed_cmds.pl new file mode 100644 index 0000000..c1b59ba --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/gene_gff3_to_bed_cmds.pl @@ -0,0 +1,28 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 gene_gff3.list\n\n"; + +my $gene_gff3_files = $ARGV[0] or die $usage; + +main: { + + open (my $fh, $gene_gff3_files) or die $!; + + while (<$fh>) { + chomp; + my $filename = $_; + + my $cmd = "gene_gff3_to_bed.pl $filename > $filename.bed"; + + print "$cmd\n"; + } + + close $fh; + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/iso_reco_analysis/gmap_to_ref.pl b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/gmap_to_ref.pl new file mode 100644 index 0000000..b002b94 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/gmap_to_ref.pl @@ -0,0 +1,45 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use File::Basename; +use Cwd; + +my $usage = "usage: $0 trin_fa.list\n\n"; + +my $trin_fa_files = $ARGV[0] or die $usage; + +my $workdir = cwd(); + + +open (my $fh, $trin_fa_files) or die "Error, cannot open file $trin_fa_files"; +while (<$fh>) { + chomp; + my $trin_fa_file = $_; + + my $outdir = dirname($trin_fa_file); + + my $cmd = "gmap -g $outdir/gene.fa $trin_fa_file -f 3 > $trin_fa_file.gff3"; + + + #&process_cmd($cmd); + print "$cmd\n"; + +} +exit(0); + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + diff --git a/99.scripts/trinity_utils/util/misc/iso_reco_analysis/notes b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/notes new file mode 100644 index 0000000..bbad6d9 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/notes @@ -0,0 +1,41 @@ +## partition the genes +~/GITHUB/trinityrnaseq/util/misc/genome_gff3_to_gene_gff3_partitions.pl genes.gff3 genes.fa 500 +find gene_contigs/ -regex ".*gene.fa" | tee fa_files.list + +## simulate reads: +sim_reads.pl fa_files.list | tee sim.cmds + + +# Trinity +~/GITHUB/trinityrnaseq/util/misc/iso_reco_analysis/run_trinity_no_LR.pl fa_files.list | tee trin.noFL.cmds + +~/GITHUB/trinityrnaseq/util/misc/iso_reco_analysis/run_trinity_WITH_LR.pl ./fa_files.list | tee trin.withFL.cmds + +# gmap the Trinity reconstructed transcripts to the gene sequence +find gene_contigs/ -regex ".*Trinity.fasta" > trin_fa.list +~/GITHUB/trinityrnaseq/util/misc/iso_reco_analysis/gmap_to_ref.pl trin_fa.list > gmap.cmds + +# cufflinks reconstruct +find gene_contigs/ -regex ".*genome.sam.coordSorted.bam" | tee bams.list +~/GITHUB/trinityrnaseq/util/misc/iso_reco_analysis/bam_to_cuff.pl bams.list | tee cuff.cmds + +# convert gff3 files to bed +find gene_contigs/ -regex ".*gene.gff3" | tee gene.gff3.list +~/GITHUB/trinityrnaseq/util/misc/iso_reco_analysis/gene_gff3_to_bed_cmds.pl gene.gff3.list > gene.gff3.list.cmds + +# convert cuff gtf to bed: +~/GITHUB/trinityrnaseq/util/misc/iso_reco_analysis/cuff_gtf_to_bed.pl cuff.list |tee cuff.list.cmds + + + +######################################## +# pull together all results for viewing. + +find bin_0/ -regex '.*gene.gff3.bed' -exec cat {} \; | sort -k1,1 -k2,2n > all_genes.gff3.bed + +find bin_0 -regex ".*trinity_WITH_LR_outdir.Trinity.fasta.gff3.bed" -exec cat {} \; | sort -k1,1 -k2,2n > trin.WITH_LR.bed + +find bin_0 -regex ".*trinity_no_LR_outdir.Trinity.fasta.gff3.bed" -exec cat {} \; | sort -k1,1 -k2,2n > trin.no_LR.bed + +find bin_0 -regex ".*transcripts.gtf.bed" -exec cat {} \; | sort -k1,1 -k2,2n > cuff_trans.bed + diff --git a/99.scripts/trinity_utils/util/misc/iso_reco_analysis/run_trinity_WITH_LR.pl b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/run_trinity_WITH_LR.pl new file mode 100644 index 0000000..2fec29e --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/run_trinity_WITH_LR.pl @@ -0,0 +1,49 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use File::Basename; +use Cwd; + +my $usage = "usage: $0 genome_fa_files.list\n\n"; + +my $genome_fa_files = $ARGV[0] or die $usage; + +my $workdir = cwd(); + + +open (my $fh, $genome_fa_files) or die "Error, cannot open file $genome_fa_files"; +while (<$fh>) { + chomp; + my $genome_file = $_; + + my $outdir = dirname($genome_file); + + if ($outdir !~ /^\./) { + $outdir = "$workdir/$outdir"; + } + + + my $cmd = "$ENV{TRINITY_HOME}/Trinity --seqType fa --single $outdir/simul.reads.fa --CPU 1 --max_memory 1G --output $outdir/trinity_WITH_LR_outdir --full_cleanup --long_reads $outdir/simul.transcriptome.cdnas --trinity_complete"; + + #&process_cmd($cmd); + print "$cmd\n"; + +} +exit(0); + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + diff --git a/99.scripts/trinity_utils/util/misc/iso_reco_analysis/run_trinity_no_LR.pl b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/run_trinity_no_LR.pl new file mode 100644 index 0000000..794f3e8 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/run_trinity_no_LR.pl @@ -0,0 +1,40 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use File::Basename; + +my $usage = "usage: $0 genome_fa_files.list\n\n"; + +my $genome_fa_files = $ARGV[0] or die $usage; + +open (my $fh, $genome_fa_files) or die "Error, cannot open file $genome_fa_files"; +while (<$fh>) { + chomp; + my $genome_file = $_; + + my $outdir = dirname($genome_file); + + my $cmd = "$ENV{TRINITY_HOME}/Trinity --seqType fa --single $outdir/simul.reads.fa --CPU 1 --max_memory 1G --output $outdir/trinity_no_LR_outdir --full_cleanup --trinity_complete"; + + #&process_cmd($cmd); + print "$cmd\n"; + +} +exit(0); + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + diff --git a/99.scripts/trinity_utils/util/misc/iso_reco_analysis/sim_reads.pl b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/sim_reads.pl new file mode 100644 index 0000000..f79a47b --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/sim_reads.pl @@ -0,0 +1,43 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use File::Basename; + +my $usage = "usage: $0 genome_fa_files.list\n\n"; + +my $genome_fa_files = $ARGV[0] or die $usage; + +open (my $fh, $genome_fa_files) or die "Error, cannot open file $genome_fa_files"; +while (<$fh>) { + chomp; + my $genome_file = $_; + + my $gff3_file = $genome_file; + $gff3_file =~ s/\.fa$/\.gff3/; + + my $outdir = dirname($gff3_file); + + my $cmd = "$ENV{TRINITY_HOME}/util/misc/simulate_reads_sam_and_fa.pl --gff3 $gff3_file --genome $genome_file --frag_length 300 --read_length 76 --SS_lib_type F --out_prefix $outdir/simul"; + + #&process_cmd($cmd); + print "$cmd\n"; + +} +exit(0); + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + diff --git a/99.scripts/trinity_utils/util/misc/iso_reco_analysis/trans_gff3_to_bed_cmds.pl b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/trans_gff3_to_bed_cmds.pl new file mode 100644 index 0000000..0877b6a --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/iso_reco_analysis/trans_gff3_to_bed_cmds.pl @@ -0,0 +1,28 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 trans_gff3.list\n\n"; + +my $trans_gff3_files = $ARGV[0] or die $usage; + +main: { + + open (my $fh, $trans_gff3_files) or die $!; + + while (<$fh>) { + chomp; + my $filename = $_; + + my $cmd = "transcript_gff3_to_bed.pl $filename > $filename.bed"; + + print "$cmd\n"; + } + + close $fh; + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/iworm_welds_to_dot.pl b/99.scripts/trinity_utils/util/misc/iworm_welds_to_dot.pl new file mode 100644 index 0000000..ef80681 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/iworm_welds_to_dot.pl @@ -0,0 +1,32 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 iworm_cluster_welds_graph.txt.sorted.wIwormNames\n\n"; + +my $iworm_welds = $ARGV[0] or die $usage; + +main: { + + open (my $fh, $iworm_welds) or die "Error, cannot open file $iworm_welds"; + + print "digraph G {\n"; + + while (<$fh>) { + chomp; + my @x = split(/\s+/); + my $iworm_A = $x[1]; + my $iworm_B = $x[4]; + $iworm_A =~ s/;/_/; + $iworm_B =~ s/;/_/; + print " $iworm_A->$iworm_B\n"; + + } + print "}\n"; + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/jaccard_sam_pair_refiner.pl b/99.scripts/trinity_utils/util/misc/jaccard_sam_pair_refiner.pl new file mode 100644 index 0000000..68b5e7e --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/jaccard_sam_pair_refiner.pl @@ -0,0 +1,138 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "\n\nusage: name_sorted_paired_reads.sam\n\n"; + +my $sam_file = $ARGV[0] or die $usage; + +main: { + + + my $prev_read_name = ""; + my $prev_scaff_name = ""; + + my @reads; + + my $sam_reader = new SAM_reader($sam_file); + while ($sam_reader->has_next()) { + + my $read = $sam_reader->get_next(); + + my $scaff_name = $read->get_scaffold_name(); + my $core_read_name = $read->get_core_read_name(); + + + if ($core_read_name ne $prev_read_name) { + + if (@reads) { + &process_pairs(@reads); + @reads = (); + } + } + + push (@reads, $read); + + + $prev_read_name = $core_read_name; + $prev_scaff_name = $scaff_name; + + + } + + &process_pairs(@reads); + + + exit(0); +} + +#### +sub process_pairs { + my (@reads) = @_; + + my %scaffold_to_reads; + foreach my $read (@reads) { + my $scaff_name = $read->get_scaffold_name(); + push (@{$scaffold_to_reads{$scaff_name}}, $read); + } + + foreach my $scaff (keys %scaffold_to_reads) { + + my @scaff_reads = @{$scaffold_to_reads{$scaff}}; + + &process_scaffold_pairs(@scaff_reads); + } + + + return; +} + + + + +sub process_scaffold_pairs { + my @reads = @_; + + my @left_reads; + my @right_reads; + + foreach my $read (@reads) { + + my $read_name = $read->get_read_name(); + + #print "processing: $read_name\n"; + + if ($read_name =~ m|/1$|) { + push (@left_reads, $read); + } + elsif ($read_name =~ m|/2$|) { + push (@right_reads, $read); + } + } + + unless (@left_reads && @right_reads) { + #print "\t** no pairs...\n"; + return; + } + + my $left_read = shift @left_reads; + my $left_aligned_pos = $left_read->get_aligned_position(); + my $scaffold_name = $left_read->get_scaffold_name(); + + my $right_read = shift @right_reads; + my $right_aligned_pos = $right_read->get_aligned_position(); + + $left_read->set_mate_scaffold_name($scaffold_name); + $left_read->set_mate_scaffold_position($right_aligned_pos); + + $right_read->set_mate_scaffold_name($scaffold_name); + $right_read->set_mate_scaffold_position($left_aligned_pos); + + $left_read->set_paired(1); + $right_read->set_paired(1); + + $left_read->set_proper_pair(1); + $right_read->set_proper_pair(1); + + $left_read->set_first_in_pair(1); + $right_read->set_second_in_pair(1); + + + print $left_read->toString() . "\n"; + print $right_read->toString() . "\n"; + + + return; +} + + + + + + diff --git a/99.scripts/trinity_utils/util/misc/join_any.pl b/99.scripts/trinity_utils/util/misc/join_any.pl new file mode 100644 index 0000000..7b64dd5 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/join_any.pl @@ -0,0 +1,51 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "\nusage: $0 token_list_file file_to_join [-v]\n\n"; + +my $token_list_file = $ARGV[0] or die $usage; +my $file_to_join = $ARGV[1] or die $usage; +my $invert_selection = $ARGV[2] || 0; + +my %tokens; +{ + open (my $fh, $token_list_file) or die "Error, cannot open file $token_list_file"; + while (<$fh>) { + while (/(\S+)/g) { + $tokens{$1} = 1; + } + } + close $fh; +} + + +open (my $fh, $file_to_join) or die "Error, cannot open file $file_to_join "; +while (<$fh>) { + my $line = $_; + chomp; + my @x = split (/\s+/); + my $found_token = 0; + foreach my $ele (@x) { + if ($tokens{$ele}) { + $found_token = 1; + last; + } + } + + if ($found_token && !$invert_selection) { + print $line; + } + elsif ($invert_selection && !$found_token) { + print $line; + } + +} +close $fh; + + +exit(0); + + + diff --git a/99.scripts/trinity_utils/util/misc/join_by_left_col.pl b/99.scripts/trinity_utils/util/misc/join_by_left_col.pl new file mode 100644 index 0000000..ac1b74e --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/join_by_left_col.pl @@ -0,0 +1,44 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 fileA fileB ...\n\n"; + +my @files = @ARGV; +unless (@files) { + die $usage; +} + +main: { + + my %data; + foreach my $file (@files) { + open (my $fh, $file) or die $!; + while (<$fh>) { + chomp; + unless (/\w/) { next; } + my ($key, $rest) = split(/\t/, $_, 2); + $rest =~ s/\t/ /g; + $data{$key}->{$file} = $rest; + } + close $fh; + } + + print join("\t", @files) . "\n"; + foreach my $acc (keys %data) { + print "$acc"; + foreach my $file (@files) { + my $val = $data{$acc}->{$file}; + unless (defined $val) { + $val = "NA"; + } + print "\t$val"; + } + print "\n"; + } + + exit(0); + +} + diff --git a/99.scripts/trinity_utils/util/misc/join_expr_vals_single_table.pl b/99.scripts/trinity_utils/util/misc/join_expr_vals_single_table.pl new file mode 100644 index 0000000..138c1a5 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/join_expr_vals_single_table.pl @@ -0,0 +1,52 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + + +my $usage = "usage: $0 (FPKM|RAW) fileA fileB ...\n\n"; + +unless (@ARGV && scalar(@ARGV) > 2 && $ARGV[0] =~ /^(FPKM|RAW)$/) { + die $usage; +} + +my $dat_type = shift @ARGV; +my @files = @ARGV; + +my %data; + +foreach my $file (@files) { + + open (my $fh, $file) or die "Error, cannot open file $file"; + while (<$fh>) { + chomp; + unless (/\w/) { next; } + if (/^\#/) { + next; + } + my @x = split(/\t/); + my $trans = $x[0]; + + my $val = ($dat_type eq "FPKM") ? $x[5] : $x[4]; + $data{$trans}->{$file} = $val; + } + close $fh; +} + + +print "#transcript\t" . join("\t", @files) . "\n"; + +foreach my $transcript (sort keys %data) { + + print $transcript; + + foreach my $file (@files) { + + my $val = $data{$transcript}->{$file} || 0; + print "\t$val"; + } + print "\n"; +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/kmer_counter.pl b/99.scripts/trinity_utils/util/misc/kmer_counter.pl new file mode 100644 index 0000000..06b1e1a --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/kmer_counter.pl @@ -0,0 +1,47 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; +use Ktree; +use Nuc_translator; + +my $usage = "usage: $0 fasta_file KmerSize [DSmode]\n\n"; + +my $fasta_file = $ARGV[0] or die $usage; +my $kmer_length = $ARGV[1] or die $usage; +my $DS_mode = $ARGV[2] || 0; + + +main: { + + my $ktree = new Ktree(); + + my $fasta_reader = new Fasta_reader($fasta_file); + while (my $seq_obj = $fasta_reader->next()) { + + my $seq = $seq_obj->get_sequence(); + my @chars = split(//, $seq); + + for (my $i = 0; $i <= length($seq) - $kmer_length; $i++) { + + my $kmer = join("", @chars[$i..($i+$kmer_length-1)]); + + $ktree->add_kmer($kmer); + + if ($DS_mode) { + $ktree->add_kmer( &reverse_complement($kmer) ); + } + } + } + + $ktree->report_kmer_counts(); + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/m8_blastclust.pl b/99.scripts/trinity_utils/util/misc/m8_blastclust.pl new file mode 100644 index 0000000..57f090a --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/m8_blastclust.pl @@ -0,0 +1,279 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling); +use Fasta_reader; +use SingleLinkageClusterer; +use File::Basename; +use Process_cmd; + +my $usage = <<__EOUSAGE__; + +######################################################################################## +# +# -I input fasta file +# --prot or --nuc protein or nucleotide +# +# -L min percent length cutoff (default: 50) +# -R the above percent length cutoff must be reciprocal (default: off) +# -P min percent identity (default: 75) +# +# -E max E-value (default: 1e-10) +# -H number of hits per entry (default: 100) +# +# --megablast use megablast (only for nuc mode) +# +# --use_m8 use an existing blast m8 file +# +######################################################################################### + +__EOUSAGE__ + + ; + + + +main: { + + my $fasta_file; + my $prot_mode = 0; + my $nuc_mode = 0; + my $min_percent_length = 50; + my $reciprocal_length_flag = 0; + my $min_percent_identity = 75; + my $Evalue = 1e-10; + my $num_hits = 100; + my $megablast_flag = 0; + my $blast_m8_file = ""; + + my $help_flag; + + + &GetOptions ( 'h' => \$help_flag, + 'I=s' => \$fasta_file, + 'prot' => \$prot_mode, + 'nuc' => \$nuc_mode, + 'L=i' => \$min_percent_length, + 'R' => \$reciprocal_length_flag, + 'P=i' => \$min_percent_identity, + 'E=f' => \$Evalue, + 'H=i' => \$num_hits, + 'megablast' => \$megablast_flag, + 'use_m8=s' => \$blast_m8_file, + + ); + + + + + if ($help_flag) { + die $usage; + } + + + unless ($fasta_file && ($prot_mode || $nuc_mode) ) { + die $usage; + } + + + + my %seq_lengths = &parse_seq_lengths($fasta_file); + + $num_hits++; # include self hit + + my $blast_outfile = $blast_m8_file; + unless ($blast_outfile) { + + $blast_outfile = join (".", "blast", $Evalue, $num_hits, "m8"); + } + + + unless (-s $blast_outfile) { + + ## prep fasta file for blast search and run blast + if ($prot_mode) { + unless (-s "$fasta_file.pin") { + my $cmd = "formatdb -i $fasta_file -p T"; + &process_cmd($cmd); + } + + my $cmd = "blastall -p blastp -d $fasta_file -i $fasta_file -m 8 -e $Evalue -v $num_hits -b $num_hits -F \'m S\' > $blast_outfile"; + &process_cmd($cmd); + + } + else { + #nuc mode + unless (-s "$fasta_file.nin") { + my $cmd = "formatdb -i $fasta_file -p F"; + &process_cmd($cmd); + } + + my $prog = ($megablast_flag) ? "megablast" : "blastall -p blastn"; + + my $cmd = "$prog -d $fasta_file -i $fasta_file -m 8 -e $Evalue -v $num_hits -b $num_hits -F \'m D\' > $blast_outfile"; + &process_cmd($cmd); + } + } + + ## assign length and percent length coverage info + my $length_outfile = "$blast_outfile.wLens"; + unless (-s $length_outfile) { + &compute_length_coverage($blast_outfile, $length_outfile, \%seq_lengths); + } + + &cluster_hits($length_outfile, $min_percent_length, $min_percent_identity, $reciprocal_length_flag); + + exit(0); + +} + +#### +sub cluster_hits { + my ($length_outfile, $min_percent_length, $min_percent_identity, $reciprocal_length_flag) = @_; + + my @pairs; + + open (my $fh, $length_outfile) or die "Error, cannot read file $length_outfile"; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my $accA = $x[0]; + my $accB = $x[1]; + my $per_ID = $x[2]; + my $length_covA = $x[13]; + my $length_covB = $x[15]; + + if ($accA eq $accB) { next; } # no self matches examined. + + if ($per_ID < $min_percent_identity) { next; } + + my $covA_requirements = ($length_covA >= $min_percent_length) ? 1:0; + my $covB_requirements = ($length_covB >= $min_percent_length) ? 1:0; + + if ($reciprocal_length_flag) { + + if ($covA_requirements && $covB_requirements) { + push (@pairs, [$accA, $accB]); + } + } + else { + if ($covA_requirements || $covB_requirements) { + push (@pairs, [$accA, $accB]); + } + } + } + close $fh; + + + my $clusters_outfile = "$length_outfile.L$min_percent_length.P$min_percent_identity.R$reciprocal_length_flag.clusters"; + open (my $ofh, ">$clusters_outfile") or die "Error, cannot write to file $clusters_outfile"; + + my @clusters = &SingleLinkageClusterer::build_clusters(@pairs); + + my %size_counter; + foreach my $cluster (@clusters) { + print $ofh join("\t", @$cluster) . "\n"; + + my $num_eles = scalar(@$cluster); + $size_counter{$num_eles}++; + + } + close $ofh; + + ## provide summary report for dist + print "#cluster_size\tcount\n"; + foreach my $cluster_size (sort {$a<=>$b} keys %size_counter) { + my $count = $size_counter{$cluster_size}; + print "$cluster_size\t$count\n"; + } + + return; + +} + + + + +#### +sub compute_length_coverage { + my ($blast_results_file, $length_outfile, $seq_lengths_href) = @_; + + open (my $ofh, ">$length_outfile") or die "Error, cannot write to $length_outfile"; + open (my $fh, "$blast_results_file") or die "Error, cannot read file $blast_results_file"; + + while (<$fh>) { + chomp; + my @x = split(/\t/); + my ($accA, $end5_A, $end3_A) = ($x[0], $x[6], $x[7]); + my ($accB, $end5_B, $end3_B) = ($x[1], $x[8], $x[9]); + + my $seq_lenA = $seq_lengths_href->{$accA} or die "Error, no sequence length for acc: $accA"; + my $seq_lenB = $seq_lengths_href->{$accB} or die "Error, no sequence length for acc: $accB"; + + my $hit_lenA = abs($end3_A - $end5_A) + 1; + my $hit_lenB = abs($end3_B - $end5_B) + 1; + + my $percent_lenA = sprintf("%.1f", $hit_lenA / $seq_lenA * 100); + my $percent_lenB = sprintf("%.1f", $hit_lenB / $seq_lenB * 100); + + print $ofh join("\t", @x, $seq_lenA, $percent_lenA, $seq_lenB, $percent_lenB) . "\n"; + + } + + close $ofh; + + return; +} + + + + + +#### +sub parse_seq_lengths { + my ($fasta_file) = @_; + + my %lengths; + + my $seq_lengths_file = basename($fasta_file) . ".seqLen"; + + if (-e $seq_lengths_file) { + open (my $fh, $seq_lengths_file) or die "Error, cannot open file $seq_lengths_file"; + while (<$fh>) { + chomp; + my ($len, $acc, @rest) = split(/\s+/); + $lengths{$acc} = $len; + } + close $fh; + + return(%lengths); + } + else { + + my $fasta_reader = new Fasta_reader($fasta_file); + + open (my $ofh, ">$seq_lengths_file") or die "Error, cannot write to file $seq_lengths_file"; + + while (my $seq_obj = $fasta_reader->next()) { + my $acc = $seq_obj->get_accession(); + my $sequence = $seq_obj->get_sequence(); + + my $len = length($sequence); + $lengths{$acc} = $len; + + print $ofh "$len\t$acc\n"; + + + } + close $ofh; + + return(%lengths); + } + +} diff --git a/99.scripts/trinity_utils/util/misc/map_gtf_transcripts_to_genome_annots.pl b/99.scripts/trinity_utils/util/misc/map_gtf_transcripts_to_genome_annots.pl new file mode 100644 index 0000000..1db13b7 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/map_gtf_transcripts_to_genome_annots.pl @@ -0,0 +1,629 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib/"); +use Gene_obj; +use GFF3_utils; +use GTF_utils; +use SAM_reader; +use SAM_entry; +use Overlap_info; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use Data::Dumper; +use Overlap_piler; + +$ENV{LC_ALL} = 'C'; + + +my $usage = <<_EOUSAGE_; + +######################################################################### +# +# *Required: + +# # Annot settings: +# +# --annot_gff3 gff3 file name +# OR +# --annot_gtf gtf file name +# +# # Transcript alignment settings: +# +# --alignment_gtf alignments in gtf format +# +# +# *Optional: +# +# # Intergenic settings: (default, off) +# +# --include_intergenic create features out of the intergenic regions. +# --min_intergenic_length minimum size of an intergenic feature to be included. +# +# # Reporting settings: +# +# --best only report the mapping that has the highest % overlap with the transcript +# --ignore_antisense only consider sense mappings +# --ignore_strandedness ignore all strand orientation, set to '?' as done for intergenic regions. +# +# --no_require_compatibility do not require compatible overlap (default: does. splice junctions must match up) +# +################################################################################################### + + +_EOUSAGE_ + + ; + + +my $annot_genes_gff3; +my $annot_genes_gtf; + +my $alignment_gtf; + +my $help_flag; + +my $VERBOSE = 0; +my $FUZZY_OVERLAP = 0; + +my $INCLUDE_INTERGENIC = 0; + + +my $MIN_INTERGENIC_LENGTH = 100; +my $MAX_MERGE_INDEL = 5; + +my $BEST_ONLY = 0; +my $IGNORE_ANTISENSE = 0; +my $IGNORE_STRANDEDNESS = 0; + +my $NO_REQUIRE_COMPATIBILITY = 0; + +&GetOptions ( 'h' => \$help_flag, + + 'annot_gff3=s' => \$annot_genes_gff3, + 'annot_gtf=s' => \$annot_genes_gtf, + + 'alignment_gtf=s' => \$alignment_gtf, + + + 'v' => \$VERBOSE, + 'FUZZY' => \$FUZZY_OVERLAP, + + 'include_intergenic' => \$INCLUDE_INTERGENIC, + 'min_intergenic_length=i' => \$MIN_INTERGENIC_LENGTH, + + 'best' => \$BEST_ONLY, + 'ignore_antisense' => \$IGNORE_ANTISENSE, + 'ignore_strandedness' => \$IGNORE_STRANDEDNESS, + + 'no_require_compatibility' => \$NO_REQUIRE_COMPATIBILITY, + ); + + +if ($help_flag) { die $usage; } + + +unless ($annot_genes_gff3 || $annot_genes_gtf) { + die $usage; +} + + +if (@ARGV) { + die "Options: @ARGV not understood"; +} + +main: { + + my $gene_obj_indexer_href = {}; + + my $contig_to_gene_list_href; + + print STDERR "-parsing gene annotations file\n"; + if ($annot_genes_gff3) { + + ## associate gene identifiers with contig id's. + $contig_to_gene_list_href = &GFF3_utils::index_GFF3_gene_objs($annot_genes_gff3, $gene_obj_indexer_href); + } + else { + # GTF mode + $contig_to_gene_list_href = >F_utils::index_GTF_gene_objs_from_GTF($annot_genes_gtf, $gene_obj_indexer_href); + } + + + ## Populate Genomic Features + + print STDERR "-organizing annotated transcript features.\n"; + + my %chr_to_features = &get_chr_to_features($contig_to_gene_list_href, $gene_obj_indexer_href); + + if ($INCLUDE_INTERGENIC) { + print STDERR "-adding intergenic features\n"; + &add_intergenic_features(\%chr_to_features); + } + + my %feature_lengths = &get_feature_lengths(\%chr_to_features); + + + ## Populate Transcript Alignment Features + my $trans_obj_indexer_href = {}; + my $contig_to_trans_list_href = >F_utils::index_GTF_gene_objs_from_GTF($alignment_gtf, $trans_obj_indexer_href); + my %trans_align_features = &get_chr_to_features($contig_to_trans_list_href, $trans_obj_indexer_href); + + + ## assign read mapping to features, including sense and antisense mappings + print STDERR "\n\n-mapping transcripts to features\n"; + my $mapped_trans_file = &map_transcript_alignments_to_genome_features(\%chr_to_features, \%trans_align_features); + + + + print STDERR "\n\n-refining mappings\n\n"; + &refine_mappings_estimate_counts($mapped_trans_file); + + + exit(0); + + +} + + +#### +sub get_chr_to_features { + my ($contig_to_gene_list_href, $gene_obj_indexer_href) = @_; + + + my %chr_to_features; + + my $total_transcripts = 0; + + foreach my $asmbl_id (sort keys %$contig_to_gene_list_href) { + + my @gene_ids = @{$contig_to_gene_list_href->{$asmbl_id}}; + + foreach my $gene_id (@gene_ids) { + my $gene_obj_ref = $gene_obj_indexer_href->{$gene_id}; + + unless (ref $gene_obj_ref) { + die "Error, no gene_obj for gene_id: $gene_id"; + } + + my $strand = $gene_obj_ref->get_orientation(); + + + my $scaffold = $asmbl_id; + + my $max_isoform_cdna_length = 0; + + foreach my $isoform ($gene_obj_ref, $gene_obj_ref->get_additional_isoforms()) { + + $total_transcripts++; + + my $isoform_id = join("::", $gene_id, $isoform->{Model_feat_name}); ## embed the gene identifier in with the transcript id, so we can tease it apart later. + my $isoform_length = 0; + + my @coordset = &get_coordset_for_isoform($isoform); + my ($lend, $rend) = ($coordset[0]->[0], $coordset[$#coordset]->[1]); + + my $length = &sum_coordset_segments(\@coordset); + + + push (@{$chr_to_features{$scaffold}}, { + acc => $isoform_id, + name => $isoform->{com_name}, + lend => $lend, + rend => $rend, + coordset => [@coordset], + strand => $strand, + length => $length, + } ); + + + } + } + } + + return(%chr_to_features); + +} + + + +#### +sub add_intergenic_features { + my ($chr_to_features_href) = @_; + + my $inter_counter = 0; + + foreach my $feature_list_aref (values %$chr_to_features_href) { + + my @coords; + foreach my $feature (@$feature_list_aref) { + my ($lend, $rend) = ($feature->{lend}, $feature->{rend}); + push (@coords, [$lend,$rend]); + } + + my @piles = &Overlap_piler::simple_coordsets_collapser(@coords); + + my $prev_rend = undef; + foreach my $pile (@piles) { + my ($curr_lend, $curr_rend) = @$pile; + if ($prev_rend) { + my $inter_lend = $prev_rend + 1; + my $inter_rend = $curr_lend - 1; + + my $inter_len = $inter_rend - $inter_lend + 1; + + if ($inter_lend < $inter_rend && $inter_len >= $MIN_INTERGENIC_LENGTH) { + $inter_counter++; + + push (@{$feature_list_aref}, { + acc => "intergeneic_region.$inter_counter", + name => "intergenic region", + lend => $inter_lend, + rend => $inter_rend, + coordset => [ [$inter_lend, $inter_rend] ], + strand => '+', + length => $inter_len, + } ); + } + + } + $prev_rend = $curr_rend; + } + + } + + + return; +} + + + +#### +sub map_transcript_alignments_to_genome_features { + my ($chr_to_features_href, $trans_align_features_href) = @_; + + ## + ## Map transcripts to genes + ## + + + + print STDERR "-examining transcript alignments, mapping to annotated features.\n"; + + + my $trans_mapping_file = "trans_align_mappings.$$.txt"; + open (my $ofh, ">$trans_mapping_file") or die $!; + + foreach my $scaffold (sort keys %$trans_align_features_href) { + + + my @chr_features; + if (exists $chr_to_features_href->{$scaffold}) { + @chr_features = sort {$a->{lend} <=> $b->{lend}} @{$chr_to_features_href->{$scaffold}}; + + } + + my @trans_features = sort {$a->{rend}<=>$b->{rend}} @{$trans_align_features_href->{$scaffold}}; + + foreach my $trans_feature (@trans_features) { + my $position_lend = $trans_feature->{lend}; + my $position_rend = $trans_feature->{rend}; + + my $trans_feature_coordset = $trans_feature->{coordset}; + my $trans_feature_strand = $trans_feature->{strand}; + + + my @container; ## holds current features. + + ## collect features that overlap + while (@chr_features && + $chr_features[0]->{lend} <= $position_rend) { + + my $feature = shift @chr_features; + if ($feature->{rend} >= $position_lend) { + + push (@container, $feature); + } + else { + # no overlap of feature with read range + # no op, feature gets tossed. + } + } + + ## purge current contained features that no longer overlap position. + @container = sort {$a->{rend}<=>$b->{rend}} @container; + while (@container && $container[0]->{rend} < $position_lend) { + shift @container; + } + + + my $mapped_read_flag = 0; + if (@container) { + + foreach my $feature (@container) { + my $acc = $feature->{acc}; + + my $strand_mapping; + if ($IGNORE_STRANDEDNESS || $feature->{strand} eq '?') { + $strand_mapping = '?'; # for intergenic, consider as sense + } + elsif ($feature->{strand} eq $trans_feature_strand) { + $strand_mapping = "SENSE"; + } + else { + $strand_mapping = "ANTI"; + } + + + if ($NO_REQUIRE_COMPATIBILITY || &Overlap_info::compatible_overlap($feature->{coordset}, $trans_feature_coordset)) { + + ## determine percent of feature aligning + my $overlapping_bases = &Overlap_info::sum_overlaps($feature->{coordset}, $trans_feature_coordset); + + if ($overlapping_bases > 0) { + + my $percent_of_trans_length_aligned = sprintf("%.2f", $overlapping_bases / $trans_feature->{length} * 100); + + print $ofh join("\t", $trans_feature->{acc}, $scaffold, $position_lend, $acc, $strand_mapping, $percent_of_trans_length_aligned) . "\n"; + + $mapped_read_flag = 1; + + } + } + + + } + + } + + + + unless ($mapped_read_flag) { + print $ofh join("\t", $trans_feature->{acc}, ".", ".", ".", ".", ".", ".") . "\n"; + } + + #if ($read_counter > 100000) { last; } ###DEBUG + } + } + + close $ofh; + + return($trans_mapping_file); +} + + +#### +sub refine_mappings_estimate_counts { + my ($trans_mapping_file) = @_; + + ## sort by read name, molecule, and position + my $read_name_sorted_file = "$trans_mapping_file.read_name_sorted"; + my $cmd = "sort -k1,1 $trans_mapping_file > $read_name_sorted_file"; + &process_cmd($cmd); + + + my @curr_read_structs; + + my %feature_to_counts; + + open (my $fh, $read_name_sorted_file) or die "Error, cannot read file $read_name_sorted_file"; + while (<$fh>) { + chomp; + my ($read_acc, $mol, $pos, $feature_acc, $sense_or_anti, $percent_mapped) = split(/\t/); + + my $read_struct = { + read_acc => $read_acc, + mol => $mol, + pos => $pos, + feature_acc => $feature_acc, + sense_or_anti => $sense_or_anti, + percent_mapped => $percent_mapped, + }; + + if (@curr_read_structs && $curr_read_structs[0]->{read_acc} ne $read_acc) { + my @features = &refine_read_mappings(@curr_read_structs); + foreach my $feature (@features) { + print join("\t", + $feature->{read_acc}, + $feature->{mol}, + $feature->{pos}, + $feature->{feature_acc}, + $feature->{sense_or_anti}, + $feature->{percent_mapped}, + ) . "\n"; + } + + @curr_read_structs = (); # reinit + } + + push (@curr_read_structs, $read_struct); + } + close $fh; + + if (@curr_read_structs) { + my @features = &refine_read_mappings(@curr_read_structs); + + foreach my $feature (@features) { + print join("\t", + $feature->{read_acc}, + $feature->{mol}, + $feature->{pos}, + $feature->{feature_acc}, + $feature->{sense_or_anti}, + $feature->{percent_mapped}, + ) . "\n"; + + } + } + + + unlink($trans_mapping_file, $read_name_sorted_file); # remove intermediate files + + +} + + + +#### +sub refine_read_mappings { + my (@curr_trans_structs) = @_; + + @curr_trans_structs = reverse sort {$a->{percent_mapped}<=>$b->{percent_mapped}} @curr_trans_structs; + + if ($IGNORE_ANTISENSE) { + @curr_trans_structs = grep { $_->{sense_or_anti} !~ /anti/i } @curr_trans_structs; + } + + unless (@curr_trans_structs) { + return (); # nothing to do + } + + if ($BEST_ONLY) { + @curr_trans_structs = shift @curr_trans_structs; + } + + + return(@curr_trans_structs); + + +} + + + +#### +sub get_coordset_for_isoform { + my ($isoform) = @_; + + my @coordsets; + + my @exons = $isoform->get_exons(); + + foreach my $exon (@exons) { + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + + push (@coordsets, [$lend, $rend]); + } + + @coordsets = sort {$a->[0]<=>$b->[0]} @coordsets; + + + return(@coordsets); +} + + +#### +sub sum_coordset_segments { + my ($coordset_aref) = @_; + + my $sum = 0; + foreach my $coordset (@$coordset_aref) { + + my ($lend, $rend) = @$coordset; + + $sum += $rend - $lend + 1; + } + + return($sum); +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + +#### +sub get_min_max_coords { + my ($read_align_coords_aref) = @_; + + my @coords; + foreach my $coordset (@$read_align_coords_aref) { + my ($lend, $rend) = @$coordset; + + push (@coords, $lend, $rend); + } + + @coords = sort {$a<=>$b} @coords; + + my $min = shift @coords; + my $max = pop @coords; + + return($min, $max); +} + +#### +sub compute_aligned_read_length { + my ($read_coords_aref) = @_; + + my $sum_len = 0; + foreach my $coordset (@$read_coords_aref) { + my ($lend, $rend) = @$coordset; + + $sum_len += abs($rend - $lend) + 1; + } + + return($sum_len); +} + + +#### +sub get_feature_lengths { + my ($chr_to_features_href) = @_; + + my %feature_lengths; + + foreach my $feature_list_aref (values %$chr_to_features_href) { + + foreach my $feature (@$feature_list_aref) { + + my $acc = $feature->{acc}; + my $len = $feature->{length}; + + $feature_lengths{$acc} = $len; + } + } + + + return(%feature_lengths); +} + + +#### +sub merge_short_indels { + my ($align_coords_aref) = @_; + + if (scalar (@$align_coords_aref) == 1) { + return($align_coords_aref); # nothing to do + } + + my @merged_coords = shift @$align_coords_aref; + + foreach my $coordset (shift @$align_coords_aref) { + my ($lend, $rend) = @$coordset; + my $prev_rend = $merged_coords[$#merged_coords]->[1]; + if (abs($lend - $prev_rend) <= $MAX_MERGE_INDEL) { + if ($rend > $prev_rend) { + $merged_coords[$#merged_coords]->[1] = $rend; + } + } + else { + # new coord segment + push (@merged_coords, $coordset); + } + } + + return(\@merged_coords); +} + + diff --git a/99.scripts/trinity_utils/util/misc/merge_RSEM_output_to_matrix.pl b/99.scripts/trinity_utils/util/misc/merge_RSEM_output_to_matrix.pl new file mode 100644 index 0000000..1122d92 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/merge_RSEM_output_to_matrix.pl @@ -0,0 +1,142 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + +use File::Basename; + +my $usage = <<__EOUSAGE__; + + +############################################################################################ +# +# Required: +# +# --rsem_files file containing a list of RSEM output files. +# (note, should be isoform or gene.results files, dont mix) +# +# --mode counts|fpkm|tpm +# +# Optional: +# +# --trans_mode_bundle_gene_id specify the transcript identifier as 'gene|trans' +# +############################################################################################ + +__EOUSAGE__ + + ; + + + +my $rsem_files_list_file; +my $mode; +my $trans_mode_bundle_gene_id = 0; +my $help_flag; + +&GetOptions( 'help|h' => \$help_flag, + + 'rsem_files=s' => \$rsem_files_list_file, + 'mode=s' => \$mode, + 'trans_mode_bundle_gene_id' => \$trans_mode_bundle_gene_id, + ); + +if ($help_flag) { + die $usage; +} +unless ($rsem_files_list_file && $mode) { + + die $usage; +} + +unless ($mode =~ /^(counts|fpkm|tpm)$/i) { + die "Error, mode $mode not recognized"; +} + + +=header_format + +0 transcript_id +1 gene_id +2 length +3 effective_length +4 expected_count +5 TPM +6 FPKM +7 IsoPct + +=cut + + +main: { + + + unless (-f $rsem_files_list_file) { die "$usage\nError: cannot open file $rsem_files_list_file"; } + my @rsem_files = `cat $rsem_files_list_file`; + chomp @rsem_files; + + my %data; + + foreach my $file (@rsem_files) { + + print STDERR "-capturing $mode from file: $file\n"; + + open (my $fh, $file) or die "Error, cannot open file $file"; + my $header = <$fh>; # ignore it + while (<$fh>) { + chomp; + my @x = split(/\t/); + my $acc = $x[0]; + + if ($trans_mode_bundle_gene_id) { + my $gene = $x[1]; + $acc = "$gene|$acc"; + } + + my $count = $x[4]; + my $tpm = $x[5]; + my $fpkm = $x[6]; + + + $data{$acc}->{$file} = ($mode =~ /counts/i) ? $count : ($mode =~ /fpkm/) ? $fpkm : $tpm; + } + close $fh; + } + + + print STDERR "\n\n* done parsing files. Now outputting matrix.\n\n"; + + my @filenames = @rsem_files; + foreach my $file (@filenames) { + $file = basename($file); + $file =~ s/-/_/g; # R doesn't like '-' in column headers + $file =~ s/\.(genes|isoforms)\.results//; + } + + + print join("\t", "", @filenames) . "\n"; + foreach my $acc (keys %data) { + + print "$acc"; + + foreach my $file (@rsem_files) { + + my $count = $data{$acc}->{$file}; + unless (defined $count) { + $count = "NA"; + } + + print "\t$count"; + + } + + print "\n"; + + } + + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/merge_blast_n_rsem_results.pl b/99.scripts/trinity_utils/util/misc/merge_blast_n_rsem_results.pl new file mode 100644 index 0000000..f5e4c9e --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/merge_blast_n_rsem_results.pl @@ -0,0 +1,60 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + +my $usage = "usage: $0 rsem.out blast.outfmt6 [transcripts.fasta]\n\n"; + +my $rsem_out = $ARGV[0] or die $usage; +my $blast_out = $ARGV[1] or die $usage; +my $transcripts_fasta = $ARGV[2]; + +main: { + + my %rsem_text; + my $rsem_header; + { + open (my $fh, $rsem_out) or die $!; + $rsem_header = <$fh>; + chomp $rsem_header; + while (<$fh>) { + chomp; + my $line = $_; + my @x = split(/\t/); + my ($gene, $trans_list) = ($x[0], $x[1]); + foreach my $ele ($gene, split(/,/, $trans_list)) { + $rsem_text{$ele} = $line; + } + } + close $fh; + } + + my %trans_seqs; + if ($transcripts_fasta) { + my $fasta_reader = new Fasta_reader($transcripts_fasta); + %trans_seqs = $fasta_reader->retrieve_all_seqs_hash($transcripts_fasta); + } + + + open (my $fh, $blast_out) or die $!; + my $header = <$fh>; + print "$rsem_header\t$header"; + while (<$fh>) { + chomp; + my $line = $_; + my @x = split(/\t/); + my $acc = $x[0]; + my $rsem_line = $rsem_text{$acc} or die "Error, no rsem text for $acc"; + print "$rsem_line\t$line"; + if (my $seq = $trans_seqs{$acc}) { + print "\t$seq"; + } + print "\n"; + } + close $fh; + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/misc/merge_replicate_bams_via_samples_file.pl b/99.scripts/trinity_utils/util/misc/merge_replicate_bams_via_samples_file.pl new file mode 100644 index 0000000..9c67cd2 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/merge_replicate_bams_via_samples_file.pl @@ -0,0 +1,40 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "\n\n\tusage: $0 samples_file dir/containing/bams\n\n"; + +my $samples_file = $ARGV[0] or die $usage; +my $bam_dir_path = $ARGV[1] or die $usage; + + +my %tissue_id_to_bams; + +open(my $fh, $samples_file) or die $!; +while(<$fh>) { + chomp; + my ($tissue_type, $sample_id, @rest) = split(/\t/); + my @bams = <$bam_dir_path/$sample_id.*.bam>; + unless (scalar(@bams) == 1) { + die "Error, cannot find bam corresping to $bam_dir_path/$sample_id.*.bam "; + } + push (@{$tissue_id_to_bams{$tissue_type}}, $bams[0]); +} +close $fh; + +foreach my $tissue_id (keys %tissue_id_to_bams) { + + my @bams = @{$tissue_id_to_bams{$tissue_id}}; + + my $cmd = "samtools merge $tissue_id.bam @bams"; + print "$cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, $cmd died with ret $ret"; + } +} + +exit(0) + + diff --git a/99.scripts/trinity_utils/util/misc/merge_rsem_n_express_for_compare.pl b/99.scripts/trinity_utils/util/misc/merge_rsem_n_express_for_compare.pl new file mode 100644 index 0000000..62ce04a --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/merge_rsem_n_express_for_compare.pl @@ -0,0 +1,87 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 rsem.file eXpress.file count|FPKM\n\n"; + +my $rsem_file = $ARGV[0] or die $usage; +my $express_file = $ARGV[1] or die $usage; +my $method = $ARGV[2] or die $usage; + +unless ($method =~ /^(count|FPKM)$/) { + die $usage; +} + +main: { + + my %data; + &add_RSEM_data(\%data, $rsem_file, $method); + &add_eXpress_data(\%data, $express_file, $method); + + + print join("\t", "", "RSEM", "eXpress") . "\n"; + foreach my $id (keys %data) { + my $rsem_val = $data{$id}->{RSEM}; + unless (defined $rsem_val) { + $rsem_val = -1; + } + my $express_val = $data{$id}->{eXpress}; + unless (defined $express_val) { + $express_val = -1; + } + + if ($rsem_val > 0 && $express_val > 0) { + $rsem_val = "NA" if $rsem_val < 0; + $express_val = "NA" if $express_val < 0; + + print join("\t", $id, $rsem_val, $express_val) . "\n"; + } + } + + exit(0); +} + +#### +sub add_RSEM_data { + my ($data_href, $rsem_file, $method) = @_; + + my $data_field = ($method eq 'count') ? 4 : 6; + + open (my $fh, $rsem_file) or die "Error, cannot open file $rsem_file"; + my $header = <$fh>; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my $id = $x[0]; + my $val = $x[$data_field]; + + $data_href->{$id}->{RSEM} = $val; + } + close $fh; + + return; +} + +#### +sub add_eXpress_data { + my ($data_href, $express_file, $method) = @_; + + my $data_field = ($method eq 'count') ? 7 : 10; + + open (my $fh, $express_file) or die "Error, cannot open file $express_file"; + my $header = <$fh>; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my $id = $x[1]; + my $val = $x[$data_field]; + + $data_href->{$id}->{eXpress} = $val; + } + close $fh; + + return; +} + + diff --git a/99.scripts/trinity_utils/util/misc/mpi_iworm_proc_contigs_to_fa.pl b/99.scripts/trinity_utils/util/misc/mpi_iworm_proc_contigs_to_fa.pl new file mode 100644 index 0000000..78536ec --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/mpi_iworm_proc_contigs_to_fa.pl @@ -0,0 +1,26 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 tmp.MPIiworm.rank.txt\n\n"; + +my $file = $ARGV[0] or die $usage; + +main: { + open (my $fh, $file) or die $!; + + my $seq_counter = 0; + while (<$fh>) { + chomp; + $seq_counter++; + + print ">s$seq_counter\n$_\n"; + + } + + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/nameSorted_SAM_to_FastQ.pl b/99.scripts/trinity_utils/util/misc/nameSorted_SAM_to_FastQ.pl new file mode 100644 index 0000000..f253388 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/nameSorted_SAM_to_FastQ.pl @@ -0,0 +1,85 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use SAM_reader; +use SAM_entry; +use Nuc_translator; + +my $usage = "usage: $0 file.sam out_prefix\n\n"; + + +my $sam_file = $ARGV[0] or die $usage; +my $out_prefix = $ARGV[1] or die $usage; + + + +main: { + + my $sam_reader = new SAM_reader($sam_file); + + my %fhs; + + while (my $sam_entry = $sam_reader->get_next()) { + + if ($sam_entry->is_query_unmapped()) { + next; + } + + my $outfh = &get_output_file($sam_entry, \%fhs); + + my $seq = $sam_entry->get_sequence(); + my $quals = $sam_entry->get_quality_scores(); + + my $strand = $sam_entry->get_query_strand(); + if ($strand eq '-') { + $seq = &reverse_complement($seq); + $quals = join("", reverse(split(//, $quals))); + } + + my $read_name = $sam_entry->reconstruct_full_read_name(); + + print $outfh join("\n", "\@$read_name", $seq, "+", $quals) . "\n"; + } + + exit(0); +} + +#### +sub get_output_file { + my ($sam_entry, $fhs_href) = @_; + + my $filename = "$out_prefix.single.fq"; + + if ($sam_entry->is_paired()) { + + if ($sam_entry->is_first_in_pair()) { + + $filename = "$out_prefix.left.fq"; + } + elsif ($sam_entry->is_second_in_pair()) { + + $filename = "$out_prefix.right.fq"; + } + else { + die "Error, read is paired but neither first or second in pair!"; + } + } + + my $fh = $fhs_href->{$filename}; + + unless ($fh) { + + open ($fh, ">$filename") or die "Error, cannot write to $filename"; + $fhs_href->{$filename} = $fh; + } + + + return($fh); +} + + diff --git a/99.scripts/trinity_utils/util/misc/nameSorted_SAM_to_paired_fastq.pl b/99.scripts/trinity_utils/util/misc/nameSorted_SAM_to_paired_fastq.pl new file mode 100644 index 0000000..6af3bc9 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/nameSorted_SAM_to_paired_fastq.pl @@ -0,0 +1,145 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use SAM_reader; +use SAM_entry; +use Nuc_translator; + +use Carp; +use Data::Dumper; + + +my $DEBUG = 0; + +my $usage = "usage: $0 file.sam out_prefix [STRICT]\n\n"; + + +my $sam_file = $ARGV[0] or die $usage; +my $out_prefix = $ARGV[1] or die $usage; +my $STRICT_FLAG = $ARGV[2] || 0; + + +my $left_fq_filename = "$out_prefix.left.fq"; +my $right_fq_filename = "$out_prefix.right.fq"; + +open(my $left_ofh, ">$left_fq_filename") or die "Error, cannot write to $left_fq_filename"; +open(my $right_ofh, ">$right_fq_filename") or die "Error, cannot write to $right_fq_filename"; + +main: { + + my $sam_reader = new SAM_reader($sam_file); + + my %fhs; + + my $prev_read_name = ""; + + my @left_entries; + my @right_entries; + + my $counter = 0; + + while (my $sam_entry = $sam_reader->get_next()) { + + my $core_read_name = $sam_entry->get_read_name(); + + print STDERR "processing $core_read_name\n" if $DEBUG; + + if (! $sam_entry->is_paired()) { + confess "ERROR, only paired reads should exist in bam file. Encountered unpaired read: " . Dumper($sam_entry); + } + + if ($prev_read_name && $core_read_name ne $prev_read_name) { + + print STDERR "\t-printing record for $core_read_name\n" if $DEBUG; + + &process_entry($prev_read_name, \@left_entries, \@right_entries); + @left_entries = (); + @right_entries = (); + + unless ($core_read_name gt $prev_read_name) { + #confess "Error, it appears the sam file is not sorted by read name, as $core_read_name ! > $prev_read_name"; + # no longer die here: different tools sort in different ways, where samtools is using some mixed string and integer sorting that's not lexicographical. + } + + $counter++; + if ($counter % 1e6 == 0) { + print STDERR "[$counter] PE fastq records written.\n"; + } + + } + + $prev_read_name = $core_read_name; + if ($sam_entry->is_first_in_pair()) { + push (@left_entries, $sam_entry); + } + elsif ($sam_entry->is_second_in_pair()) { + push (@right_entries, $sam_entry); + } + else { + confess "Error, cannot determine sam entry as first or second in pair: " . Dumper($sam_entry); + } + } + + ## get last one + &process_entry($prev_read_name, \@left_entries, \@right_entries); + + + exit(0); +} + + +#### +sub process_entry { + my ($core_read_name, $left_entries_aref, $right_entries_aref) = @_; + + unless (@$left_entries_aref) { + print STDERR "WARNING: $core_read_name is missing first-read entry in sam file. skipping...\n"; + if ($STRICT_FLAG) { + confess "ERROR: $core_read_name is missing first-read entry in sam file, STRICT mode enabled"; + } + return; + } + unless (@$right_entries_aref) { + print STDERR "WARNING: $core_read_name is missing second-read entry in sam file. skipping...\n"; + if ($STRICT_FLAG) { + confess "ERROR: $core_read_name is missing second-read entry in sam file. STRICT mode enabled"; + } + + return; + } + + my $left_sam_entry = $left_entries_aref->[0]; + &print_fastq_record($left_ofh, $left_sam_entry); + + my $right_sam_entry = $right_entries_aref->[0]; + &print_fastq_record($right_ofh, $right_sam_entry); + + return; +} + +#### +sub print_fastq_record { + my ($ofh, $sam_entry) = @_; + + + my $read_name = $sam_entry->reconstruct_full_read_name(); + + my $seq = $sam_entry->get_sequence(); + my $quals = $sam_entry->get_quality_scores(); + + my $strand = $sam_entry->get_query_strand(); + if ($strand eq '-') { + $seq = &reverse_complement($seq); + $quals = join("", reverse(split(//, $quals))); + } + + print $ofh join("\n", "\@$read_name", $seq, "+", $quals) . "\n"; + + return; +} + diff --git a/99.scripts/trinity_utils/util/misc/omp_iworm_thread_contigs_to_fa.pl b/99.scripts/trinity_utils/util/misc/omp_iworm_thread_contigs_to_fa.pl new file mode 100644 index 0000000..81c8490 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/omp_iworm_thread_contigs_to_fa.pl @@ -0,0 +1,35 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 tmp.iworm.thread.txt\n\n"; + +my $file = $ARGV[0] or die $usage; + +main: { + open (my $fh, $file) or die $!; + + my $seq_counter = 0; + reader: + while (1) { + + my @vals; + for (1..4) { + my $line = <$fh>; + if (eof($fh)) { last reader; } + chomp $line; + push (@vals, $line); + } + + $seq_counter++; + + my $seq = pop @vals; + print ">s$seq_counter\n$seq\n"; + + } + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/organize_data_table_by_trinity_component.pl b/99.scripts/trinity_utils/util/misc/organize_data_table_by_trinity_component.pl new file mode 100644 index 0000000..c2017bd --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/organize_data_table_by_trinity_component.pl @@ -0,0 +1,32 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + + +my %data; +while (<>) { + if (/^\#/) { + print; + next; + } + unless (/\w/) { next; } + + my $line = $_; + if (/^(comp\d+_c\d+)/) { + my $comp = $1; + $data{$comp} .= $line; + } + else { + die "Error, cannot decode component identity from $line"; + } +} + +foreach my $component (keys %data) { + + print $data{$component} . "\n"; + +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/pair_up_fastq_files_1_2.pl b/99.scripts/trinity_utils/util/misc/pair_up_fastq_files_1_2.pl new file mode 100644 index 0000000..0391585 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/pair_up_fastq_files_1_2.pl @@ -0,0 +1,30 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Cwd; + +my $curr_dir = cwd(); + +foreach my $file (<*_1.*fastq*>, <*_1*.fq*>) { + + $file =~ /^(\S+)_1\.*/ or die "Error, cannot decipher filename $file"; + my $core = $1; + + my $right_fq = $file; + $right_fq =~ s/_1\./_2\./; + + unless (-s $right_fq) { + die "Error, cannot find Right.fq file corresponding to $file"; + } + + print join("\t", $core, "$curr_dir/$file", "$curr_dir/$right_fq") . "\n"; + +} + + +exit(0); + + + diff --git a/99.scripts/trinity_utils/util/misc/pair_up_fastq_files_LeftRight.pl b/99.scripts/trinity_utils/util/misc/pair_up_fastq_files_LeftRight.pl new file mode 100644 index 0000000..bf1ee0b --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/pair_up_fastq_files_LeftRight.pl @@ -0,0 +1,32 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Cwd; + +my $curr_dir = cwd(); + +foreach my $file (<*.Left.fq*>, <*.left.fq*>, <*_Left.fq*>, <*_left.fq*>) { + + $file =~ /^(\S+)[\._]Left.fq/i or die "Error, cannot decipher filename $file"; + my $core = $1; + + my $right_fq = $file; + $right_fq =~ s/([\._])Left/${1}Right/; + $right_fq =~ s/([\._])left/${1}right/; + + + unless (-s $right_fq) { + die "Error, cannot find Right.fq file corresponding to $file"; + } + + print join("\t", $core, "$curr_dir/$file", "$curr_dir/$right_fq") . "\n"; + +} + + +exit(0); + + + diff --git a/99.scripts/trinity_utils/util/misc/pair_up_fastq_files_R1_R2.pl b/99.scripts/trinity_utils/util/misc/pair_up_fastq_files_R1_R2.pl new file mode 100644 index 0000000..03815fa --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/pair_up_fastq_files_R1_R2.pl @@ -0,0 +1,30 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Cwd; + +my $curr_dir = cwd(); + +foreach my $file (<*_R1*.fastq*>, <*_R1*.fq*>, <*.R1*.fastq*>, <*.R1*.fq*>) { + + $file =~ /^(\S+)[\._]R1[\._].*f(ast)?q/ or die "Error, cannot decipher filename $file"; + my $core = $1; + + my $right_fq = $file; + $right_fq =~ s/([\._])R1([\._])/$1R2$2/ or die "Error, cannot convert R1 to R2 in $right_fq"; + + unless (-s $right_fq) { + die "Error, cannot find Right.fq file corresponding to $file"; + } + + print join("\t", $core, "$curr_dir/$file", "$curr_dir/$right_fq") . "\n"; + +} + + +exit(0); + + + diff --git a/99.scripts/trinity_utils/util/misc/pairwise_kmer_content_comparer.pl b/99.scripts/trinity_utils/util/misc/pairwise_kmer_content_comparer.pl new file mode 100644 index 0000000..38bf39a --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/pairwise_kmer_content_comparer.pl @@ -0,0 +1,98 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + +my $usage = "usage: $0 file.fasta [kmer_length=25]\n\n"; + +my $fasta_file = $ARGV[0] or die $usage; +my $kmer_size = $ARGV[1] || 25; + + +main: { + + my $fasta_reader = new Fasta_reader($fasta_file); + my %trans_seqs = $fasta_reader->retrieve_all_seqs_hash(); + + my %acc_to_kmers = &get_kmers(\%trans_seqs); + + my @accs = keys %acc_to_kmers; + + + for (my $i = 0; $i < $#accs; $i++) { + + my $acc_i = $accs[$i]; + my $kmers_href_i = $acc_to_kmers{$acc_i}; + + + for (my $j = $i + 1; $j <= $#accs; $j++) { + + my $acc_j = $accs[$j]; + my $kmers_href_j = $acc_to_kmers{$acc_j}; + + if (&have_kmer_in_common($kmers_href_i, $kmers_href_j)) { + + print join("\t", $acc_i, $acc_j) . "\n"; + } + } + + } + + exit(0); + +} + +#### +sub get_kmers { + my ($trans_seqs_href) = @_; + + my %acc_to_kmers; + + my $counter = 0; + foreach my $acc (keys %$trans_seqs_href) { + $counter++; + print STDERR "-getting kmers for $counter\n"; + + + my $trans_seq = $trans_seqs_href->{$acc}; + + my %kmers = &extract_kmers($trans_seq); + $acc_to_kmers{$acc} = \%kmers; + } + + return(%acc_to_kmers); +} + +#### +sub extract_kmers { + my ($seq) = @_; + + my %kmers; + + for (my $i = 0; $i <= length($seq) - $kmer_size; $i++) { + my $kmer = substr($seq, $i, $kmer_size); + + $kmers{$kmer} = 1; + } + + return(%kmers); +} + +#### +sub have_kmer_in_common { + my ($kmers_i, $kmers_j) = @_; + + foreach my $kmer (keys %$kmers_i) { + + if (exists $kmers_j->{$kmer}) { + return(1); + } + } + + return(0); +} + diff --git a/99.scripts/trinity_utils/util/misc/plot_ExN50_statistic.Rscript b/99.scripts/trinity_utils/util/misc/plot_ExN50_statistic.Rscript new file mode 100644 index 0000000..e380960 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/plot_ExN50_statistic.Rscript @@ -0,0 +1,43 @@ +#!/usr/bin/env Rscript + +args<-commandArgs(TRUE) + +if (length(args) == 0) { + stop("\n\n\tusage: plot_ExN50_statistic.Rscript sampleA.ExN50.stats [ sampleB.ExN50.stats ... ] \n\n\n") +} + + +library(tidyverse) + +alldata = NULL + +for (i in 1:length(args)) { + filename = args[i] + + message(sprintf("parsing: %s", filename)) + data = read.table(filename, header=T, row.names=NULL) + + data$sample = filename + + if (is.null(alldata)) { + alldata <- data + } else { + alldata <- rbind(alldata, data) + } +} + +if (length(args) == 1) { + pdf_filename = paste0(basename(args[1]), ".ExN50_plot.pdf") +} else { + pdf_filename = "ExN50_plot.pdf" +} +pdf(pdf_filename) + +p = alldata %>% filter(Ex >= 30) %>% ggplot(aes(x=Ex, y=ExN50, color=sample)) + geom_line() + xlim(c(30,100)) + +plot(p) + +write(cat("ExN50 data plotted as:", pdf_filename), stderr()) + +quit(save = "no", status = 0, runLast = FALSE) + diff --git a/99.scripts/trinity_utils/util/misc/plot_expressed_gene_dist.pl b/99.scripts/trinity_utils/util/misc/plot_expressed_gene_dist.pl new file mode 100644 index 0000000..4c28d21 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/plot_expressed_gene_dist.pl @@ -0,0 +1,38 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; + +my $usage = "usage: $0 RSEM.isoforms.fpkm\n\n"; + +my $fpkm_file = $ARGV[0] or die $usage; + +my $Rscript = "$fpkm_file.R"; +open (my $ofh, ">$Rscript"); +print $ofh "source(\"$FindBin::RealBin/R/expression_analysis_lib.R\")\n"; +print $ofh "pdf(\"$fpkm_file.genes_vs_minFPKM.pdf\")\n"; +print $ofh "plot_expressed_gene_counts(\"$fpkm_file\", title=\"expressed transcripts vs. min FPKM\", fpkm_range=seq(0,5,0.01), outfile=\"$fpkm_file.genes_vs_minFPKM.dat\")\n"; +print $ofh "dev.off()\n"; +close $ofh; + +&process_cmd("R --no-save --no-restore --no-site-file --no-init-file -q < $Rscript"); + + +exit(0); + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} diff --git a/99.scripts/trinity_utils/util/misc/plot_strand_specificity_dist_by_quantile.Rscript b/99.scripts/trinity_utils/util/misc/plot_strand_specificity_dist_by_quantile.Rscript new file mode 100644 index 0000000..7503801 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/plot_strand_specificity_dist_by_quantile.Rscript @@ -0,0 +1,180 @@ +#!/usr/bin/env Rscript + +args<-commandArgs(TRUE) + +if (length(args) == 0) { + stop("\n\nusage: plot_strand_specificity_dist_by_quantile.Rscript ss_analysis.dat\n\n") +} + + +dat_filename = args[1] + +library(vioplot) + + + +## Vioplot2 function from: http://stackoverflow.com/questions/22410606/violin-plot-with-list-input +vioplot2<-function (x, ..., range = 1.5, h = NULL, ylim = NULL, names = NULL, + horizontal = FALSE, col = "magenta", border = "black", lty = 1, + lwd = 1, rectCol = "black", colMed = "white", pchMed = 19, + at, add = FALSE, wex = 1, drawRect = TRUE) +{ + if(!is.list(x)){ + datas <- list(x, ...) + } else{ + datas<-x + } + n <- length(datas) + if (missing(at)) + at <- 1:n + upper <- vector(mode = "numeric", length = n) + lower <- vector(mode = "numeric", length = n) + q1 <- vector(mode = "numeric", length = n) + q3 <- vector(mode = "numeric", length = n) + med <- vector(mode = "numeric", length = n) + base <- vector(mode = "list", length = n) + height <- vector(mode = "list", length = n) + baserange <- c(Inf, -Inf) + args <- list(display = "none") + if (!(is.null(h))) + args <- c(args, h = h) + for (i in 1:n) { + data <- datas[[i]] + data.min <- min(data) + data.max <- max(data) + q1[i] <- quantile(data, 0.25) + q3[i] <- quantile(data, 0.75) + med[i] <- median(data) + iqd <- q3[i] - q1[i] + upper[i] <- min(q3[i] + range * iqd, data.max) + lower[i] <- max(q1[i] - range * iqd, data.min) + est.xlim <- c(min(lower[i], data.min), max(upper[i], + data.max)) + smout <- do.call("sm.density", c(list(data, xlim = est.xlim), + args)) + hscale <- 0.4/max(smout$estimate) * wex + base[[i]] <- smout$eval.points + height[[i]] <- smout$estimate * hscale + t <- range(base[[i]]) + baserange[1] <- min(baserange[1], t[1]) + baserange[2] <- max(baserange[2], t[2]) + } + if (!add) { + xlim <- if (n == 1) + at + c(-0.5, 0.5) + else range(at) + min(diff(at))/2 * c(-1, 1) + if (is.null(ylim)) { + ylim <- baserange + } + } + if (is.null(names)) { + label <- 1:n + } + else { + label <- names + } + boxwidth <- 0.05 * wex + if (!add) + plot.new() + if (!horizontal) { + if (!add) { + plot.window(xlim = xlim, ylim = ylim) + axis(2) + axis(1, at = at, label = label) + } + box() + for (i in 1:n) { + polygon(c(at[i] - height[[i]], rev(at[i] + height[[i]])), + c(base[[i]], rev(base[[i]])), col = col, border = border, + lty = lty, lwd = lwd) + if (drawRect) { + lines(at[c(i, i)], c(lower[i], upper[i]), lwd = lwd, + lty = lty) + rect(at[i] - boxwidth/2, q1[i], at[i] + boxwidth/2, + q3[i], col = rectCol) + points(at[i], med[i], pch = pchMed, col = colMed) + } + } + } + else { + if (!add) { + plot.window(xlim = ylim, ylim = xlim) + axis(1) + axis(2, at = at, label = label) + } + box() + for (i in 1:n) { + polygon(c(base[[i]], rev(base[[i]])), c(at[i] - height[[i]], + rev(at[i] + height[[i]])), col = col, border = border, + lty = lty, lwd = lwd) + if (drawRect) { + lines(c(lower[i], upper[i]), at[c(i, i)], lwd = lwd, + lty = lty) + rect(q1[i], at[i] - boxwidth/2, q3[i], at[i] + + boxwidth/2, col = rectCol) + points(med[i], at[i], pch = pchMed, col = colMed) + } + } + } + invisible(list(upper = upper, lower = lower, median = med, + q1 = q1, q3 = q3)) +} + + + + + +data = read.table(dat_filename, header=T, row.names=1, com='', sep='\t') + +data = data[rev(order(data$total_reads)),] # just to be sure they're in descending order of total counts. + +sum_reads = sum(data$total_reads) +c = cumsum(data$total_reads) + +cuts = list() +cutnames = c() + +data$diff_ratio = data$diff_ratio + rnorm(nrow(data))/100 + + +cut5 = data$diff_ratio[c <= 0.05 * sum_reads] +if (length(cut5) > 0) { + cuts = append(cuts, list(cut5)) + cutnames = c(cutnames, '5%') +} + +cut10 = data$diff_ratio[c <= 0.10 * sum_reads] +if (length(cut10) > 0) { + cuts = append(cuts, list(cut10)) + cutnames = c(cutnames, '10%') +} + +cut25 = data$diff_ratio[c <= 0.25 * sum_reads] +if (length(cut25) > 0) { + cuts = append(cuts, list(cut25)) + cutnames = c(cutnames, '25%') +} + +cut50 = data$diff_ratio[c <= 0.5 * sum_reads] +if (length(cut50) > 0) { + cuts = append(cuts, list(cut50)) + cutnames = c(cutnames, '50%') +} + +cut100 = data$diff_ratio +cuts = append(cuts, list(cut100)) +cutnames = c(cutnames, '100%') + + + +plot_filename = paste0(dat_filename, ".vioplot.pdf") +pdf(plot_filename) + + +vioplot2(cuts, names=cutnames, ylim=c(-1,1)) + + +dev.off() + +quit(save = "no", status = 0, runLast = FALSE) + diff --git a/99.scripts/trinity_utils/util/misc/print.pl b/99.scripts/trinity_utils/util/misc/print.pl new file mode 100644 index 0000000..68b1cb3 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/print.pl @@ -0,0 +1,50 @@ +#!/usr/bin/env perl + +unless (@params = @ARGV) { + die "usage: $0 [-t|-s] \n"; +} + +$delimeter = pop @params; + +if ($delimeter eq '-s' || $delimeter eq '-t') { + if ($delimeter eq '-s') { + $delimeter = '\s+'; + } elsif ($delimeter eq '-t') { + $delimeter = '\t'; + } + pop @ARGV; +} else { + $delimeter = '\t'; +} + +if ($ARGV[0] =~ /(^\d+)\-(\d+)/) { + #print "$1\t$2\n"; + @array = ($1 .. $2); + +} else { + @array = @ARGV; +} + + +foreach $entry (@array) { + $here{$entry} = 1; +} + + +while () { + chomp; + #my $tab = 0; + @columns = split (/$delimeter/, $_); + my $output = ""; + for ($i = 0; $i <= $#columns; $i++) { + #if ($tab) { print "\t";} + if ($here{$i}) { + #print $columns[$i]; + $output .= "$columns[$i]\t"; + #$tab = 1; + } + } + $output =~ s/\s+$//; # remove trailing ws. + print "$output\n"; +} + diff --git a/99.scripts/trinity_utils/util/misc/print_kmers.pl b/99.scripts/trinity_utils/util/misc/print_kmers.pl new file mode 100644 index 0000000..7d6b958 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/print_kmers.pl @@ -0,0 +1,31 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + +my $usage = "usage: $0 file.fa [kmer_length=25]\n\n"; + +my $fa_file = $ARGV[0] or die $usage; +my $kmer_length = $ARGV[1] || 25; + +main: { + + my $fasta_reader = new Fasta_reader($fa_file); + while (my $seq_obj = $fasta_reader->next()) { + + my $sequence = $seq_obj->get_sequence(); + + for (my $i = 0; $i < length($sequence) - $kmer_length + 1; $i++) { + + my $kmer = substr($sequence, $i, $kmer_length); + + print "$kmer\n"; + } + } + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/misc/process_GMAP_alignments_gff3_chimeras_ok.pl b/99.scripts/trinity_utils/util/misc/process_GMAP_alignments_gff3_chimeras_ok.pl new file mode 100644 index 0000000..bb23a19 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/process_GMAP_alignments_gff3_chimeras_ok.pl @@ -0,0 +1,167 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use File::Basename; +use Cwd; + +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + + +my $usage = <<__EOUSAGE__; + +###################################################################### +# +# Required: +# --genome target genome to align to +# --transcripts cdna sequences to align +# +# Optional: +# -N number of top hits (default: 1) +# -I max intron length +# --CPU number of threads (default: 2) +# --no_chimera do not report chimeric alignmetnts +# --SAM output in SAM format +# --gtf gene structure annotations in gtf format (for genome building) +# --splice_assist splice assist mode (introns or splicesites) +# +# +####################################################################### + + +__EOUSAGE__ + + ; + + +my ($genome, $transcriptDB, $max_intron); +my $CPU = 2; + +my $help_flag; + +my $number_top_hits = 1; +my $no_chimera_flag = 0; +my $SAM_flag = 0; +my $gtf_file = 0; +my $splice_assist = ""; + +&GetOptions( 'h' => \$help_flag, + 'genome=s' => \$genome, + 'transcripts=s' => \$transcriptDB, + 'I=i' => \$max_intron, + 'CPU=i' => \$CPU, + 'N=i' => \$number_top_hits, + 'no_chimera' => \$no_chimera_flag, + 'SAM' => \$SAM_flag, + 'gtf=s' => \$gtf_file, + 'splice_assist=s' => \$splice_assist, + ); + + +if ($help_flag) { + die $usage; +} + + +unless ($genome && $transcriptDB) { + die $usage; +} + +if ($splice_assist) { + unless ($splice_assist =~ /^(introns|splicesites)$/) { + die "Error, splice_assist $splice_assist not recognized as an option"; + } +} + + +main: { + + my $genomeName = basename($genome); + my $genomeDir = $genomeName . ".gmap"; + + my $genomeBaseDir = dirname($genome); + + my $cwd = cwd(); + + unless (-d "$genomeBaseDir/$genomeDir") { + + #my $cmd = "gmap_build -D $genomeBaseDir -d $genomeBaseDir/$genomeDir -k 13 $genome >&2"; + my $cmd = "gmap_build -D $genomeBaseDir -d $genomeDir -k 13 $genome >&2"; + &process_cmd($cmd); + + + if ($gtf_file) { + &build_intron_n_splice_info_files($genomeBaseDir, $genomeDir, $gtf_file); + } + + } + + + ## run GMAP + + my $num_gmap_top_hits = $number_top_hits; + if ((! $no_chimera_flag) && $num_gmap_top_hits == 1) { + $num_gmap_top_hits = 0; # reports two hits if chimera with this setting. + } + + my $format = ($SAM_flag) ? "samse" : "3"; + + my $cmd = "gmap -D $genomeBaseDir -d $genomeDir $transcriptDB -f $format -n $num_gmap_top_hits -x 50 -t $CPU -B 5 "; + if ($max_intron) { + $cmd .= " --intronlength=$max_intron "; + } + + if ($splice_assist) { + $cmd .= " -m ref_${splice_assist}.iit"; + } + + &process_cmd($cmd); + + exit(0); +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + #return; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret ($ret)"; + } + + return; +} + +#### +sub build_intron_n_splice_info_files { + my ($genomeBaseDir, $genomeDir, $gtf_file) = @_; + + my $genome_maps_dir = "$genomeBaseDir/$genomeDir/$genomeDir.maps"; + + my $introns_iit_file = "$genome_maps_dir/ref_introns.iit"; + + my $cmd = "gtf_introns < $gtf_file | iit_store -o $introns_iit_file"; + &process_cmd($cmd); + + if (! -e $introns_iit_file) { + die "Error, no introns $introns_iit_file"; + } + + my $splicesites_iit_file = "$genome_maps_dir/ref_splicesites.iit"; + + $cmd = "gtf_splicesites < $gtf_file | iit_store -o $splicesites_iit_file"; + &process_cmd($cmd); + + if (! -e $splicesites_iit_file) { + die "Error, no splicesites $splicesites_iit_file"; + } + + return; +} + diff --git a/99.scripts/trinity_utils/util/misc/process_minimap2_alignments.pl b/99.scripts/trinity_utils/util/misc/process_minimap2_alignments.pl new file mode 100644 index 0000000..3c3fa24 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/process_minimap2_alignments.pl @@ -0,0 +1,183 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use File::Basename; +use Cwd; + +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + + +my $usage = <<__EOUSAGE__; + +###################################################################### +# +# Required: +# --genome target genome to align to +# --transcripts cdna sequences to align +# +# Optional: +# --gtf gene structure annotations in gtf format +# --CPU number of threads (default: 2) +# -o|--output bam output filename (default: basename(transcripts).mm2.bam) +# +# -I|--max_intron_length maximum intron length (default: 100000) +# +# --incl_out_gff3 include gff3 formatted output file for alignments. +# --allow_secondary allow secondary alignments (default secondary=no) +# +# --eqx include --eqx flag +# --cs include long format via --cs w/ minimap2 +# --hq pacbio CCS reads (--splice:hq for mm2) +# +####################################################################### + + +__EOUSAGE__ + + ; + +my $genome; +my $transcripts; +my $gtf; +my $CPU = 2; + +my $help_flag; +my $output; +my $max_intron_length = 100000; +my $incl_out_gff3; +my $allow_secondary = 0; +my $include_cs_flag = 0; +my $include_eqx_flag = 0; +my $include_hq_flag = 0; + +&GetOptions( 'h' => \$help_flag, + 'genome=s' => \$genome, + 'transcripts=s' => \$transcripts, + 'gtf=s' => \$gtf, + 'CPU=i' => \$CPU, + 'o|output=s' => \$output, + 'I|max_intron_length=i' => \$max_intron_length, + 'incl_out_gff3' => \$incl_out_gff3, + 'allow_secondary' => \$allow_secondary, + 'cs' => \$include_cs_flag, + 'hq' => \$include_hq_flag, + ); + + +if ($help_flag) { + die $usage; +} + + +unless ($genome && $transcripts) { + die $usage; +} + + +my $include_cs_param = ""; +my $include_cs_token = ""; +if ($include_cs_flag) { + $include_cs_param = "--cs"; + $include_cs_token = ".cs"; +} + +my $include_eqx_param = ""; +my $include_eqx_token = ""; +if ($include_eqx_flag) { + $include_eqx_param = "--eqx"; + $include_eqx_token = ".eqx"; +} + + +my $include_hq_param = ""; +my $include_hq_token = ""; +if ($include_hq_flag) { + $include_hq_param = ":hq"; + $include_hq_token = ".hq"; +} + + +unless ($output) { + $output = basename($transcripts) . ".mm2${include_cs_token}${include_eqx_token}${include_hq_token}.bam"; +} + + + + +main: { + + + my $genomeBaseDir = dirname($genome); + my $genomeName = basename($genome); + my $mm2_idx = "$genomeBaseDir/$genomeName" . ".mm2"; + + my $cwd = cwd(); + + my $splice_file = "$mm2_idx.splice.bed"; + + unless (-e $mm2_idx) { + + my $cmd = "minimap2 -d $mm2_idx $genome"; + &process_cmd($cmd); + } + + + if ($gtf && ! -s $splice_file) { + my $cmd = "paftools.js gff2bed $gtf > $splice_file"; + &process_cmd($cmd); + } + + + ## run minimap2 + + my $splice_param = ""; + if ($splice_file) { + $splice_param = "--junc-bed $splice_file"; + } + + my $secondary = ($allow_secondary) ? "" : "--secondary=no"; + + my $cmd = "minimap2 --sam-hit-only -ax splice$include_hq_param $splice_param $secondary -t $CPU -u b -G $max_intron_length $include_cs_param $include_eqx_param $mm2_idx $transcripts > $output.tmp.sam"; + &process_cmd($cmd); + + $cmd = "samtools view -Sb -T $genome $output.tmp.sam -o $output.tmp.unsorted.bam"; + &process_cmd($cmd); + + $cmd = "samtools sort $output.tmp.unsorted.bam -o $output"; + &process_cmd($cmd); + + $cmd = "samtools index $output"; + #&process_cmd($cmd); + `$cmd`; # ignore error that occurs if file is too big. + + if ($incl_out_gff3) { + $cmd = "$FindBin::Bin/SAM_to_gff3.minimap2.pl $output > $output.gff3"; + &process_cmd($cmd); + } + + unlink("$output.tmp.sam", "$output.tmp.unsorted.bam"); + + exit(0); +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + #return; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret ($ret)"; + } + + return; +} + + + diff --git a/99.scripts/trinity_utils/util/misc/prop_pair_sam_refiner.pl b/99.scripts/trinity_utils/util/misc/prop_pair_sam_refiner.pl new file mode 100644 index 0000000..5fae6c9 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/prop_pair_sam_refiner.pl @@ -0,0 +1,103 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "\n\nusage: name_sorted_paired_reads.sam\n\n"; + +my $sam_file = $ARGV[0] or die $usage; + +main: { + + + my $prev_read_name = ""; + my $prev_scaff_name = ""; + + my @reads; + + my $sam_reader = new SAM_reader($sam_file); + while ($sam_reader->has_next()) { + + + my $read = $sam_reader->get_next(); + + my $scaff_name = $read->get_scaffold_name(); + my $core_read_name = $read->get_core_read_name(); + + if ($scaff_name ne $prev_scaff_name || $core_read_name ne $prev_read_name) { + + if (@reads) { + &process_pairs(@reads); + @reads = (); + } + } + + push (@reads, $read); + + + $prev_read_name = $core_read_name; + $prev_scaff_name = $scaff_name; + + + } + + &process_pairs(@reads); + + + exit(0); +} + +#### +sub process_pairs { + my (@reads) = @_; + + + my @left_reads; + my @right_reads; + + foreach my $read (@reads) { + if ($read->is_first_in_pair()) { + push (@left_reads, $read); + } + elsif ($read->is_second_in_pair()) { + push (@right_reads, $read); + } + } + + unless (@left_reads && @right_reads) { + die "Error, dont have pairs!"; + } + + my $left_read = shift @left_reads; + my $aligned_pos = $left_read->get_aligned_position(); + + my $right_read = undef; + while ((! defined($right_read)) && @right_reads) { + my $read = shift @right_reads; + if ($read->get_mate_scaffold_position() == $aligned_pos) { + $right_read = $read; + last; + } + } + + unless ($left_read && $right_read) { + die "Error, couldn't match pairs."; + } + + print $left_read->toString() . "\n"; + print $right_read->toString() . "\n"; + + + return; +} + + + + + + diff --git a/99.scripts/trinity_utils/util/misc/randomly_mutate_seqs.pl b/99.scripts/trinity_utils/util/misc/randomly_mutate_seqs.pl new file mode 100644 index 0000000..0ffeca7 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/randomly_mutate_seqs.pl @@ -0,0 +1,188 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use Getopt::Long qw(:config posix_default no_ignore_case bundling pass_through); +use lib "$ENV{TRINITY_HOME}/PerlLib"; +use Fasta_reader; +use List::Util qw (shuffle); +#use Math::Random; + + +my $usage = <<__EOUSAGE__; + +################################################################## +# +# --fasta fasta filename +# +# --subst_rate rate of single base substitutions +# +# --insert_rate rate of insertion +# +# --insert_size default: 1 +# +# --delete_rate rate for deletions +# +# --delete_size default: 1 +# +# * note, all rates must be 0 <= x <= 0.25 +# +################################################################### + + +__EOUSAGE__ + + ; + +my $fasta_file; +my $subst_rate = 0; +my $insert_rate = 0; +my $insert_size = 1; +my $delete_rate = 0; +my $delete_size = 1; + + +&GetOptions ( + 'fasta=s' => \$fasta_file, + 'subst_rate=f' => \$subst_rate, + + 'insert_rate=f' => \$insert_rate, + 'insert_size=i' => \$insert_size, + + 'delete_rate=f' => \$delete_rate, + 'delete_size=i' => \$delete_size, + + ); + + +unless ($fasta_file) { + die $usage; +} +unless ($subst_rate || $insert_rate || $delete_rate) { + die $usage; +} + + +foreach my $info_aref ( ['subst_rate', $subst_rate ], + ['insert_rate', $insert_rate ], + ['delete_rate', $delete_rate ] ) { + + my ($rate_type, $val) = @$info_aref; + if ($val > 0.25) { + die "Error, --$rate_type $val exceeds max val of 0.25 "; + } + + +} + + +main: { + + my $fasta_reader = new Fasta_reader($fasta_file); + + my %fasta_seqs = $fasta_reader->retrieve_all_seqs_hash(); + + foreach my $acc (keys %fasta_seqs) { + + + my $seq = $fasta_seqs{$acc}; + + my @seqarray = &convert_to_seqarray($seq); + + my %seen; + + + if ($subst_rate) { + @seqarray = &mutate_seq(\@seqarray, \%seen, $subst_rate, 'substitution'); + } + if ($insert_rate) { + @seqarray = &mutate_seq(\@seqarray, \%seen, $insert_rate, 'insertion'); + } + if ($delete_rate) { + @seqarray = &mutate_seq(\@seqarray, \%seen, $delete_rate, 'deletion'); + } + + my $mutated_seq = join("", @seqarray); + print ">$acc mutated\n$mutated_seq\n"; + + } + + + + exit(0); + +} + +#### +sub convert_to_seqarray { + my ($seq) = @_; + + my @chars = split(//, uc $seq); + + return(@chars); +} + +#### +sub mutate_seq { + my ($seqarray_aref, $seen_href, $mut_rate, $mut_type) = @_; + + my $seqlen = $#$seqarray_aref + 1; + + my $num_mutations = int($mut_rate * $seqlen + 0.5); + + for (1..$num_mutations) { + + my $pos = -1; + do { + $pos = int(rand($seqlen)); + + } while ($seen_href->{$pos}); + + $seen_href->{$pos} = 1; + + ## Substitutions + if ($mut_type eq 'substitution') { + + my $char = $seqarray_aref->[$pos]; + + my @mut_chars = grep { $_ ne $char } qw(G A T C); + + my $mut_base = $mut_chars[ int(rand(3)) ]; + + $seqarray_aref->[$pos] = lc $mut_base; + } + + ## Deletions + elsif ($mut_type eq 'deletion') { + for (my $i = $pos; $i < $pos + $delete_size && $i < $seqlen; $i++) { + + $seqarray_aref->[$i] = ""; + $seen_href->{$i} = 1; + } + } + + ## Insertions + elsif ($mut_type eq 'insertion') { + + ## create insertion + my $insertion_seq = ""; + my @bases = qw(G A T C); + for (my $i = 0; $i < $insert_size; $i++) { + my $base = $bases[int(rand(4))]; + $insertion_seq .= $base; + } + $seqarray_aref->[$pos] .= lc $insertion_seq; + } + else { + confess "Error, do not understand mutation type: $mut_type "; + } + + } + + + return (@$seqarray_aref); + +} + + diff --git a/99.scripts/trinity_utils/util/misc/randomly_sample_PE_fastq.pl b/99.scripts/trinity_utils/util/misc/randomly_sample_PE_fastq.pl new file mode 100644 index 0000000..d4b4829 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/randomly_sample_PE_fastq.pl @@ -0,0 +1,111 @@ +#!/usr/bin/env perl + +## util/fastQ_rand_subset.pl +## Purpose: Extracts out a specific number of random reads from an +## input file, making sure that left and right ends of the selected +## reads are paired +## Usage: $0 left.fq right.fq num_entries +## + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fastq_reader; +use File::Basename; +use List::Util qw(shuffle); +use Data::Dumper; + +my $usage = "usage: $0 left.fq right.fq num_entries num_total_records\n\n"; + +my $left_fq = $ARGV[0] or die $usage; +my $right_fq = $ARGV[1] or die $usage; +my $num_entries = $ARGV[2] or die $usage; +my $num_total_records = $ARGV[3] or die $usage; + + +main: { + + srand(); + my @selected_indices = &get_random_indices($num_entries, $num_total_records); + + if (scalar(@selected_indices) != $num_entries) { + die "Error, get_random_indices returned " . scalar(@selected_indices) . " instead of $num_total_records"; + } + else { + print STDERR "-selected $num_entries indices. Now outputting selected records\n"; + } + + my %selected = map { + $_ => 1 } @selected_indices; + + @selected_indices = (); + + print STDERR "Selecting $num_entries entries..."; + + my $left_fq_reader = new Fastq_reader($left_fq) or die("unable to open $left_fq for input"); + my $right_fq_reader = new Fastq_reader($right_fq) or die("unable to open $right_fq for input");; + + my $num_M_entries = $num_entries/1e6; + $num_M_entries .= "M"; + my $base_left_fq = basename($left_fq); + my $base_right_fq = basename($right_fq); + + + open (my $left_ofh, ">$base_left_fq.$num_M_entries.fq") or die $!; + open (my $right_ofh, ">$base_right_fq.$num_M_entries.fq") or die $!; + + my $counter = 0; + while (my $left_entry = $left_fq_reader->next()) { + my $right_entry = $right_fq_reader->next(); + + unless ($left_entry && $right_entry) { + die "Error, didn't retrieve both left and right entries from file ($left_entry, $right_entry) "; + } + unless ($left_entry->get_core_read_name() eq $right_entry->get_core_read_name()) { + die "Error, core read names don't match: " + . "Left: " . $left_entry->get_core_read_name() . "\n" + . "Right: " . $right_entry->get_core_read_name() . "\n"; + } + + if ($selected{$counter}) { + + print $left_ofh $left_entry->get_fastq_record(); + + print $right_ofh $right_entry->get_fastq_record(); + + delete($selected{$counter}); + + } + + $counter++; + if ($counter % 100000 == 0) { + print STDERR "\r[$counter] "; + } + } + + print STDERR "\n\ndone.\n"; + + close $left_ofh; + close $right_ofh; + + if (%selected) { + die "Error, missing indices: " . Dumper(\%selected); + } + else { + print STDERR "-all records located and output.\n"; + } + + exit(0); +} + + + +#### +sub get_random_indices { + my ($num_entries, $num_total_records) = @_; + + return( (shuffle 0..$num_total_records)[0..($num_entries-1)]); + +} + diff --git a/99.scripts/trinity_utils/util/misc/remove_cntrl_chars.pl b/99.scripts/trinity_utils/util/misc/remove_cntrl_chars.pl new file mode 100644 index 0000000..a769ccb --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/remove_cntrl_chars.pl @@ -0,0 +1,8 @@ +#!/usr/bin/env perl + +use strict; + +while () { + tr/\t\n\000-\037\177-\377/\t\n /d; + print; +} diff --git a/99.scripts/trinity_utils/util/misc/rename_fasta_accessions_using_Trinotate_annot_mappings.pl b/99.scripts/trinity_utils/util/misc/rename_fasta_accessions_using_Trinotate_annot_mappings.pl new file mode 100644 index 0000000..8999a22 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/rename_fasta_accessions_using_Trinotate_annot_mappings.pl @@ -0,0 +1,58 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = <<__EOUSAGE__; + +############################################### +# +# Usage: $0 Trinity.fasta new_feature_id_mapping.txt +# +# The 'new_feature_id_mapping.txt' file has the format: +# +# current_identifier new_identifier +# .... +# +# +# Only those entries with new names listed will be updated, the rest stay unchanged. +# +# +################################################# + + +__EOUSAGE__ + + ; + +my $trinity_fasta_file = $ARGV[0] or die $usage; +my $new_feature_mappings = $ARGV[1] or die $usage; + +main: { + + my %new_ids; + { + open (my $fh, $new_feature_mappings) or die $!; + while (<$fh>) { + chomp; + my ($old_name, $new_name) = split(/\t/); + $new_ids{$old_name} = $new_name; + } + close $fh; + } + + open (my $fh, $trinity_fasta_file) or die $!; + while (my $line = <$fh>) { + if ($line =~ /^>(\S+)/) { + my $acc = $1; + if (my $new_id = $new_ids{$acc}) { + $line =~ s/^>/^>$new_id /; + #print $line; + } + } + print $line; + } + + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/misc/row_to_column.pl b/99.scripts/trinity_utils/util/misc/row_to_column.pl new file mode 100644 index 0000000..aa53753 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/row_to_column.pl @@ -0,0 +1,23 @@ +#!/usr/bin/env perl + +use strict; + +my $delimeter = ($ARGV[0] eq '-s') ? '\s+' : '\t'; + +while () { + + unless (/\w/) { + print "0\n\n"; + next; + } + + chomp; + my $x = 0; + my @x = split (/$delimeter/); + foreach my $word (@x) { + print "$x\t$word\n"; + $x++; + } + print "\n"; + #last; +} diff --git a/99.scripts/trinity_utils/util/misc/run_DETONATE.pl b/99.scripts/trinity_utils/util/misc/run_DETONATE.pl new file mode 100644 index 0000000..6473e61 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_DETONATE.pl @@ -0,0 +1,92 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Carp; +use Getopt::Long qw(:config posix_default no_ignore_case bundling pass_through); + +my $help_flag; + +my $CPU = 4; +my $output_dir = "DetonateData"; + +my $species_opts = "mouse|human|fission_yeast"; + +my $usage = <<__EOUSAGE__; + +####################################################################################### +# +# (note, must set env var DETONATE_HOME to installation directory of DETONATE software) +# +# --reads if paired-end, list as comma-delimited: "left.fq,right.fq" +# +# --target target fasta file (eg. Trinity.fasta) +# +# --frag_len frag length. If SE data, it's the read length. +# +# --species $species_opts (param files provided at: rsem-eval/true_transcript_length_distribution/) +# +# optional: +# +# --SS_lib_type R, F, RF, or FR +# +# --threads number of threads to use in multi-threading (default: $CPU) +# +# --output_dir default: $output_dir +# +######################################################################################## + + +__EOUSAGE__ + + ; + + +my $reads; +my $target; +my $frag_len; +my $SS_lib_type; +my $species; + +&GetOptions ( 'h' => \$help_flag, + + 'reads=s' => \$reads, + 'target=s' => \$target, + 'frag_len=i' => \$frag_len, + 'SS_lib_type=s' => \$SS_lib_type, + 'threads=i' => \$CPU, + 'output_dir=s' => \$output_dir, + 'species=s' => \$species, + ); + + +unless ($reads && $target && $frag_len && $species) { + die $usage; +} + + +main: { + + my $detonate_home_dir = $ENV{DETONATE_HOME} or die "Error, must set env var DETONATE_HOME to its installation directory"; + + unless (defined($species) && $species =~ /$species_opts/) { + die "Error, species $species not supported, only $species_opts"; + } + + my $cmd = "$detonate_home_dir/rsem-eval/rsem-eval-calculate-score -p $CPU " + . " --transcript-length-parameters $detonate_home_dir/rsem-eval/true_transcript_length_distribution/$species.txt " + . " $reads $target $output_dir $frag_len "; + + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + else { + print STDERR "Done.\n"; + } + exit(0); + +} diff --git a/99.scripts/trinity_utils/util/misc/run_GSNAP.pl b/99.scripts/trinity_utils/util/misc/run_GSNAP.pl new file mode 100644 index 0000000..03a99ca --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_GSNAP.pl @@ -0,0 +1,173 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use File::Basename; +use Cwd; + +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + + +my $usage = <<__EOUSAGE__; + +###################################################################### +# +# Required: +# --genome target genome to align to +# and +# --reads fastq files. If pairs, indicate both in quotes, ie. "left.fq right.fq" +# or +# --samples_file samples.txt file (format: sample_name(tab)left.fq(tab)right.fq) +# Optional: +# -N number of top hits (default: 1) +# -I max intron length (default: 1000000) +# -G GTF file for incorporating reference splice site info. +# --CPU number of threads (default: 2) +# --out_prefix output prefix (default: gsnap) +# --no_sarray skip the sarray in the gmap-build +# --proper_pairs_only require proper pairing of reads +# +####################################################################### + + +__EOUSAGE__ + + ; + + +my ($genome, $reads); + +my $max_intron = 1000000; +my $CPU = 2; + +my $help_flag; + +my $num_top_hits = 1; +my $out_prefix = "gsnap"; +my $gtf_file; +my $no_sarray = ""; +my $proper_pairs_only_flag = 0; +my $samples_file; + +&GetOptions( 'h' => \$help_flag, + 'genome=s' => \$genome, + 'reads=s' => \$reads, + 'I=i' => \$max_intron, + 'CPU=i' => \$CPU, + 'N=i' => \$num_top_hits, + 'out_prefix=s' => \$out_prefix, + 'G=s' => \$gtf_file, + 'no_sarray' => \$no_sarray, + 'proper_pairs_only' => \$proper_pairs_only_flag, + + 'samples_file=s' => \$samples_file, + + ); + + +unless ($genome && ($reads || $samples_file)) { + die $usage; +} + +if ($no_sarray) { + $no_sarray = "--no-sarray"; +} + +main: { + + my $genomeName = basename($genome); + my $genomeDir = $genomeName . ".gmap"; + + my $genomeBaseDir = dirname($genome); + + my $cwd = cwd(); + + unless (-d "$genomeBaseDir/$genomeDir") { + + my $cmd = "gmap_build -D $genomeBaseDir -d $genomeDir -T $genomeBaseDir -k 13 $no_sarray $genome >&2"; + &process_cmd($cmd); + } + + my $splice_file; + my $splice_param = ""; + + if ($gtf_file) { + $splice_file = "$gtf_file.gsnap.splice"; + if (! -s $splice_file) { + # create one. + my $cmd = "gtf_splicesites < $gtf_file > $splice_file"; + &process_cmd($cmd); + + $cmd = "iit_store -o $splice_file.iit < $splice_file"; + &process_cmd($cmd); + } + + $splice_param = "--use-splicing=$splice_file.iit"; + } + + + ## run GMAP + + my $gsnap_use_sarray = ($no_sarray) ? "--use-sarray=0" : ""; + + my @reads_files; + if ($samples_file) { + open (my $fh, $samples_file) or die $!; + while (<$fh>) { + chomp; + my ($condition, $sample_name, @read_paths) = split(/\s+/); + my $read_paths_str = join(" ", @read_paths); + push (@reads_files, [$sample_name, $read_paths_str]); + } + close $fh; + } + else { + @reads_files = [$out_prefix, $reads]; + } + + foreach my $read_set_aref (@reads_files) { + + my ($out_prefix, $reads) = @$read_set_aref; + + if ($reads =~ /\.gz$/) { + $reads .= " --gunzip"; + } + + my $require_proper_pairs = ""; + if ($proper_pairs_only_flag) { + $require_proper_pairs = " -f 2 "; + } + + my $cmd = "bash -c \"set -o pipefail && gsnap -D $genomeBaseDir -d $genomeDir -A sam -N 1 -w $max_intron $gsnap_use_sarray -n $num_top_hits -t $CPU $reads $splice_param @ARGV | samtools view -bS -F 4 $require_proper_pairs - | samtools sort -@ $CPU - -o $out_prefix.cSorted.bam \""; + &process_cmd($cmd); + + if (-s "$out_prefix.cSorted.bam") { + $cmd = "samtools index $out_prefix.cSorted.bam"; + &process_cmd($cmd); + } + } + + + exit(0); +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + #return; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret ($ret)"; + } + + return; +} + + + diff --git a/99.scripts/trinity_utils/util/misc/run_HISAT.pl b/99.scripts/trinity_utils/util/misc/run_HISAT.pl new file mode 100644 index 0000000..e2af631 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_HISAT.pl @@ -0,0 +1,221 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib("$FindBin::RealBin/../../PerlLib"); +use Pipeliner; +use File::Basename; +use Cwd; + +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + +our $HISAT_HOME; + +BEGIN { + + if ($ENV{HISAT_HOME}) { + $HISAT_HOME = $ENV{HISAT_HOME}; + } + else { + my $hisat_prog = `sh -c "command -v hisat"`; + if ($hisat_prog) { + chomp $hisat_prog; + $HISAT_HOME = dirname($hisat_prog); + } + else { + die "Error, cannot find hisat in PATH setting"; + } + } +} + + + +my $usage = <<__EOUSAGE__; + +###################################################################### +# +# Required: +# --genome target genome to align to +# --reads fastq files. If pairs, indicate both in quotes, ie. "left.fq right.fq" +# +# Optional: +# -N max number of alignments to report. (default: 1) +# -G GTF file for incorporating reference splice site info. +# --CPU number of threads (default: 2) +# --out_prefix output prefix (default: hisat) +# --run_as_single_reads if paired, run as single reads +# +####################################################################### + +Include additional hisat arguments as additional command line parameters. + +######################################################################## + +__EOUSAGE__ + + ; + + +my ($genome, $reads); + +my $CPU = 2; + +my $help_flag; + +my $out_prefix = "hisat"; +my $gtf_file; +my $run_as_single_flag = 0; +my $num_top_hits = 1; + +&GetOptions( 'h' => \$help_flag, + 'genome=s' => \$genome, + 'reads=s' => \$reads, + 'CPU=i' => \$CPU, + 'out_prefix=s' => \$out_prefix, + 'G=s' => \$gtf_file, + 'run_as_single_reads' => \$run_as_single_flag, + 'N=i' => \$num_top_hits, + ); + + + +if ($help_flag) { + die $usage; +} + +unless ($genome && $reads) { + die $usage; +} + + +main: { + + my $hisat_index = "$genome.hisat.idx"; + if (! -s "$hisat_index.1.bt2") { + ## build hisat index + + my $cmd = "$HISAT_HOME/hisat-build $genome $hisat_index"; + &process_cmd($cmd); + } + + + my $splice_incl = ""; + + if ($gtf_file) { + + my $gtf_splice = "$gtf_file.hisat.splice"; + + unless (-s $gtf_splice) { + my $cmd = "$HISAT_HOME/extract_splice_sites.py $gtf_file > $gtf_file.hisat.splice"; + &process_cmd($cmd); + } + + $splice_incl = " --known-splicesite-infile $gtf_splice "; + } + + ## run HISAT + + $reads = &add_zcat_fifo_and_add_hisat_params($reads); + + my $top_hits_count = ""; + if ($num_top_hits > 1) { + $top_hits_count = " -k $num_top_hits "; + } + + my @tmpfiles; + + my $pipeliner = new Pipeliner(-verbose => 1); + + my $cmd = "bash -c \"set -o pipefail && $HISAT_HOME/hisat -x $hisat_index -q $reads $splice_incl -p $CPU $top_hits_count @ARGV | gzip -c > $out_prefix.sam.gz\" "; + + $pipeliner->add_commands( new Command($cmd, "$out_prefix.sam.gz.ok") ); + push (@tmpfiles, "$out_prefix.sam.gz"); + + $cmd = "bash -c \"set -o pipefail && gunzip -c $out_prefix.sam.gz | samtools view -@ $CPU -F 4 -Sb -o $out_prefix.bam \""; + $pipeliner->add_commands( new Command($cmd, "$out_prefix.bam.ok") ); + push (@tmpfiles, "$out_prefix.bam"); + + + $cmd = "samtools sort -@ $CPU $out_prefix.bam -o $out_prefix.cSorted.bam"; + $pipeliner->add_commands( new Command($cmd, "$out_prefix.cSorted.bam.ok") ); + + + $pipeliner->run(); + + + if (-s "$out_prefix.cSorted.bam") { + $cmd = "samtools index $out_prefix.cSorted.bam"; + &process_cmd($cmd); + } + + unlink(@tmpfiles); + + exit(0); +} + + +#### +sub add_zcat_fifo_and_add_hisat_params { + my ($reads) = @_; + + $reads =~ s/^\s+|\s+$//g; + + my @adj_reads_list; + + my $counter = 0; + my @read_files = split(/\s+/, $reads); + + my @updated_read_filenames; + + foreach my $reads_file (@read_files) { + + $counter++; + + if ($reads_file =~ /\.gz$/) { + $reads_file = "<(zcat $reads_file)"; + } + + push (@updated_read_filenames, $reads_file); + + # add decoration + $reads_file = (scalar(@read_files) == 2) ? "-$counter $reads_file" : "-U $reads_file"; + + push (@adj_reads_list, $reads_file); + } + + if ($run_as_single_flag) { + return("-U " . join(",", @updated_read_filenames)); + } + else { + + + my $adj_reads = join(" ", @adj_reads_list); + + return($adj_reads); + } +} + + + + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + #return; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret ($ret)"; + } + + return; +} + + + diff --git a/99.scripts/trinity_utils/util/misc/run_HISAT2_via_samples_file.pl b/99.scripts/trinity_utils/util/misc/run_HISAT2_via_samples_file.pl new file mode 100644 index 0000000..6a6f700 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_HISAT2_via_samples_file.pl @@ -0,0 +1,156 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use Process_cmd; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use Pipeliner; +use File::Basename; + +my $CPU = 2; + +my $usage = <<__EOUSAGE; + +############################################################ +# +# Required: +# +# --genome target genome.fasta file +# +# --samples_file Trinity samples file +# +# Optional: +# +# --gtf annotation in gtf format +# +# --CPU multithreading (default: $CPU) +# +# --nameSorted sorts bam by read name +# +########################################################### + + +__EOUSAGE + + ; + + + + +my $help_flag; +my $genome_fa; +my $annotation_gtf; +my $samples_file; +my $nameSorted; + +&GetOptions ( 'h' => \$help_flag, + 'genome=s' => \$genome_fa, + 'gtf=s' => \$annotation_gtf, + 'samples_file=s' => \$samples_file, + 'CPU=i' => \$CPU, + 'nameSorted' => \$nameSorted, + ); + +if ($help_flag) { die $usage; } + +unless ($genome_fa && $samples_file) { + die $usage; +} + +$genome_fa = &Pipeliner::ensure_full_path($genome_fa); +$samples_file = &Pipeliner::ensure_full_path($samples_file); +$annotation_gtf = &Pipeliner::ensure_full_path($annotation_gtf) if $annotation_gtf; + + + +main: { + + my @read_sets = &parse_samples_file($samples_file); + + ## align reads to the mini-genome using hisat2 + + ########################### + # first, build genome index + + my $pipeliner = new Pipeliner(-verbose => 1); + + if ($annotation_gtf) { + + $pipeliner->add_commands(new Command("hisat2_extract_splice_sites.py $annotation_gtf > $annotation_gtf.ss", + "$annotation_gtf.ss.ok")); + + $pipeliner->add_commands(new Command("hisat2_extract_exons.py $annotation_gtf > $annotation_gtf.exons", + "$annotation_gtf.exons.ok")); + + $pipeliner->add_commands(new Command("hisat2-build --exon $annotation_gtf.exons --ss $annotation_gtf.ss -p $CPU $genome_fa $genome_fa", + "$genome_fa.hisat2.build.ok")); + + $pipeliner->run(); + } + else { + + $pipeliner->add_commands(new Command("hisat2-build -p $CPU $genome_fa $genome_fa", + "$genome_fa.hisat2.nogtf.build.ok")); + + $pipeliner->run(); + } + + ##################### + ## now run alignments + + my $aln_checkpoints_dir = "hisat2_aln_chkpts." . basename($genome_fa); + unless (-d $aln_checkpoints_dir) { + mkdir($aln_checkpoints_dir) or die "Error, cannot mkdir $aln_checkpoints_dir"; + } + + my $sorted_opt = ""; + my $sorted_token = "c"; + + if ($nameSorted) { + $sorted_opt = "-n"; + $sorted_token = "n"; + } + + foreach my $read_set_aref (@read_sets) { + my ($sample_id, $left_fq, $right_fq) = @$read_set_aref; + + + my $bamfile = "$sample_id.${sorted_token}Sorted.hisat2." . basename($genome_fa) . ".bam"; + + $pipeliner->add_commands(new Command("bash -c \"set -eof pipefail; hisat2 --dta -x $genome_fa -p $CPU -1 $left_fq -2 $right_fq | samtools view -Sb -F 4 | samtools sort $sorted_opt -o $bamfile \" ", + "$aln_checkpoints_dir/$bamfile.ok")); + + } + + $pipeliner->run(); + + exit(0); + +} + + + +#### +sub parse_samples_file { + my ($samples_file) = @_; + + my @samples; + + open(my $fh, $samples_file) or die "Error, cannot open file $samples_file"; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my ($cond, $rep, $fq_a, $fq_b) = @x; + + $fq_a = &Pipeliner::ensure_full_path($fq_a); + $fq_b = &Pipeliner::ensure_full_path($fq_b) if $fq_b; + + push (@samples, [$rep, $fq_a, $fq_b]); + } + close $fh; + + return (@samples); +} + diff --git a/99.scripts/trinity_utils/util/misc/run_HiCpipe_bowtie.pl b/99.scripts/trinity_utils/util/misc/run_HiCpipe_bowtie.pl new file mode 100644 index 0000000..baa6411 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_HiCpipe_bowtie.pl @@ -0,0 +1,52 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; + + +my $usage = "usage: $0 genome.fasta left.fq right.fq [output_dir]\n\n"; + +my $genome_file = $ARGV[0] or die $usage; +my $left_fq_file = $ARGV[1] or die $usage; +my $right_fq_file = $ARGV[2] or die $usage; +my $output_dir = $ARGV[3] || "bowtie.$$.dir"; + +while ($output_dir =~ m|/$|) { + chop $output_dir; +} + + +main: { + + ## run bowtie + my $cmd = "$FindBin::RealBin/../alignReads.pl --target $genome_file --left $left_fq_file --right $right_fq_file " + . " --seqType fq --aligner bowtie -o $output_dir --max_dist_between_pairs 900000000 --no_rsem --retain_intermediate_files " + . " -- -a -m 1 --best --strata -p 4 --chunkmbs 512 "; + &process_cmd($cmd) unless (-s "$output_dir/$output_dir.nameSorted.sam"); + + $cmd = "$FindBin::RealBin/HiCpipe_nameSortedSam_to_raw.pl $output_dir/$output_dir.nameSorted.sam > $output_dir/$output_dir.raw"; + &process_cmd($cmd) unless (-s "$output_dir/$output_dir.raw"); + + + exit(0); + + +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} diff --git a/99.scripts/trinity_utils/util/misc/run_STAR.pl b/99.scripts/trinity_utils/util/misc/run_STAR.pl new file mode 100644 index 0000000..9dd04f9 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_STAR.pl @@ -0,0 +1,236 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib("$FindBin::RealBin/../../PerlLib"); +use Pipeliner; +use File::Basename; +use Cwd; + +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + + +my $usage = <<__EOUSAGE__; + +###################################################################### +# +# Required: +# --genome target genome to align to +# --gtf|G GTF file for incorporating reference splice site info. (recommended) +# +# --reads fastq files. If pairs, indicate both in quotes, ie. "left.fq right.fq" +# +# Optional: +# --CPU number of threads (default: 2) +# --out_prefix output prefix (default: star) +# --out_dir output directory (default: current working directory) +# --star_path full path to the STAR program to use. +# --patch genomic targets to patch the genome fasta with. +# --chim_search include Chimeric.junction outputs +# --max_intron max intron length (and PE gap size) +# --one_pass do one pass alignment instead of two-pass +# +####################################################################### + + +__EOUSAGE__ + + ; + + +my ($genome, $reads); + +my $CPU = 2; + +my $help_flag; + +my $out_prefix = "star"; +my $gtf_file; +my $out_dir; +my $ADV = 0; + +my $star_path = "STAR"; +my $patch; +my $chim_search; +my $max_intron; +my $one_pass = 0; + +&GetOptions( 'h' => \$help_flag, + 'genome=s' => \$genome, + 'reads=s' => \$reads, + 'CPU=i' => \$CPU, + 'out_prefix=s' => \$out_prefix, + 'gtf|G=s' => \$gtf_file, + 'out_dir=s' => \$out_dir, + 'ADV' => \$ADV, + 'star_path=s' => \$star_path, + 'patch=s' => \$patch, + 'chim_search' => \$chim_search, + "max_intron=i" => \$max_intron, + 'one_pass' => \$one_pass, + ); + + +unless ($genome && $reads) { + die $usage; +} + +if ($help_flag) { + die $usage; +} + +if (@ARGV) { + die "Error, cannot recognize opts: @ARGV"; +} + + +my $star_prog = `sh -c "command -v $star_path"`; +chomp $star_prog; +unless ($star_prog =~ /\w/) { + die "Error, cannot locate STAR program. Be sure it's in your PATH setting. "; +} + + +main: { + + ## ensure all full paths + $genome = &Pipeliner::ensure_full_path($genome); + $gtf_file = &Pipeliner::ensure_full_path($gtf_file) if $gtf_file; + + my @read_files = split(/\s+/, $reads); + foreach my $read_file (@read_files) { + if ($read_file) { + $read_file = &Pipeliner::ensure_full_path($read_file); + } + } + $reads = join(" ", @read_files); + + if ($out_dir) { + unless (-d $out_dir) { + mkdir $out_dir or die "Error, cannot mkdir $out_dir"; + } + chdir $out_dir or die "Error, cannot cd to $out_dir"; + } + + + my $star_index = "$genome.star.idx"; + if (! -e "$star_index/build.ok") { + ## build star index + unless (-d $star_index) { + mkdir($star_index) or die "Error, cannot mkdir $star_index"; + } + + + my $cmd = "$star_prog --runThreadN $CPU --runMode genomeGenerate --genomeDir $star_index " + . " --genomeFastaFiles $genome " + . " --limitGenomeGenerateRAM 40419136213 "; + if ($gtf_file) { + $cmd .= " --sjdbGTFfile $gtf_file " + . " --sjdbOverhang 100 "; + + } + + &process_cmd($cmd); + + &process_cmd("touch $star_index/build.ok"); + + } + + + ## run STAR + + my @tmpfiles; + + my $pipeliner = new Pipeliner(-verbose => 1); + + my $pass_mode = ($one_pass) ? "None" : "Basic"; + + my $cmd = "$star_prog " + . " --runThreadN $CPU " + . " --genomeDir $star_index " + . " --outSAMtype BAM SortedByCoordinate " + . " --runMode alignReads " + . " --readFilesIn $reads " + . " --twopassMode $pass_mode " + . " --alignSJDBoverhangMin 10 " + . " --outSAMstrandField intronMotif " + . " --outSAMunmapped Within " + . " --outReadsUnmapped Fastx " + . " --alignInsertionFlush Right " + . " --alignSplicedMateMapLminOverLmate 0 " + . " --alignSplicedMateMapLmin 30 " + . " --alignSJstitchMismatchNmax 5 -1 5 5 " #which allows for up to 5 mismatches for non-canonical GC/AG, and AT/AC junctions, and any number of mismatches for canonical junctions (the default values 0 -1 0 0 replicate the old behavior (from AlexD) + . " --peOverlapNbasesMin 12 " + . " --peOverlapMMp 0.1 " + . " --limitBAMsortRAM 20000000000"; + + + if ($max_intron) { + + $cmd .= " --alignMatesGapMax $max_intron " + . " --alignIntronMax $max_intron "; + } + + + if ($chim_search) { + $cmd .= " --chimJunctionOverhangMin 8 " + . " --chimOutJunctionFormat 1 " + . " --chimSegmentMin 12 " + . " --chimSegmentReadGapMax parameter 3 " + . " --chimMultimapNmax 20 " + . " --chimOutType Junctions WithinBAM " + . " --chimScoreJunctionNonGTAG -4 " + . " --chimNonchimScoreDropMin 10 " + . " --chimMultimapScoreRange 10 "; + } + + if ($patch) { + $cmd .= " --genomeFastaFiles $patch "; + } + + + + if ($reads =~ /\.gz$/) { + $cmd .= " --readFilesCommand 'gunzip -c' "; + } + + $pipeliner->add_commands( new Command($cmd, "star_align.ok") ); + + + + my $bam_outfile = "Aligned.sortedByCoord.out.bam"; + my $renamed_bam_outfile = "$out_prefix.sortedByCoord.out.bam"; + $pipeliner->add_commands( new Command("mv $bam_outfile $renamed_bam_outfile", "$renamed_bam_outfile.ok") ); + + + $pipeliner->add_commands( new Command("samtools index $renamed_bam_outfile", "$renamed_bam_outfile.bai.ok") ); + + + $pipeliner->run(); + + + exit(0); +} + + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + #return; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret ($ret)"; + } + + return; +} + + + diff --git a/99.scripts/trinity_utils/util/misc/run_STAR_via_samples_file.pl b/99.scripts/trinity_utils/util/misc/run_STAR_via_samples_file.pl new file mode 100644 index 0000000..5cf5646 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_STAR_via_samples_file.pl @@ -0,0 +1,258 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib("$FindBin::RealBin/../../PerlLib"); +use Pipeliner; +use File::Basename; +use Cwd; +use List::Util qw(min); + +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + + +my $usage = <<__EOUSAGE__; + +###################################################################### +# +# Required: +# --genome target genome to align to +# --samples_file trinity samples file +# +# Optional +# --gtf annotations in gtf format +# --CPU number of threads (default: 2) +# --nameSorted sort bam by name instead of coordinate +# +# --max_intron maximum intron length +# --join_bio_reps search all bio replicates together instead of separately +# +####################################################################### + + +__EOUSAGE__ + + ; + + +my ($genome); +my $samples_file; + +my $CPU = 2; + +my $help_flag; +my $gtf_file; +my $nameSorted; + +my $join_bio_reps_flag = 0; + +my $max_intron_length; + + +&GetOptions( 'h' => \$help_flag, + 'genome=s' => \$genome, + 'samples_file=s' => \$samples_file, + 'CPU=i' => \$CPU, + 'gtf=s' => \$gtf_file, + 'nameSorted' => \$nameSorted, + + 'max_intron=i' => \$max_intron_length, + + 'join_bio_reps' => \$join_bio_reps_flag, + ); + + +unless ($genome && $samples_file) { + die $usage; +} + +if ($help_flag) { + die $usage; +} + +if (@ARGV) { + die "Error, cannot recognize opts: @ARGV"; +} + + +my $star_prog = `sh -c "command -v STAR"`; +chomp $star_prog; +unless ($star_prog =~ /\w/) { + die "Error, cannot locate STAR program. Be sure it's in your PATH setting. "; +} + + +main: { + + ## ensure all full paths + $genome = &Pipeliner::ensure_full_path($genome); + $gtf_file = &Pipeliner::ensure_full_path($gtf_file) if $gtf_file; + + + my $num_contigs = `grep '>' $genome | wc -l`; + chomp $num_contigs; + $num_contigs = int($num_contigs); + unless ($num_contigs > 0) { + die "Error, couldn't determine the number of contigs in genome: $genome ... shouldn't happen. "; + } + + my $genomeChrBinNbits = min(18, int(log((-s $genome) / $num_contigs) / log(2) + 0.5) ); + my $genomeSAindexNbases = min(14, int(log(-s $genome)/log(2)/2 - 1 + 0.5)); + + + my %sample_read_sets = &parse_samples_file($samples_file); + + my $pipeliner = new Pipeliner(-verbose => 1); + my $star_index = "$genome.star.idx"; + my $star_index_chkpt = "$star_index/build.ok"; + ## build star index + unless (-d $star_index) { + mkdir($star_index) or die "Error, cannot mkdir $star_index"; + } + + my $cmd = "$star_prog --runThreadN $CPU --runMode genomeGenerate --genomeDir $star_index " + . " --genomeFastaFiles $genome " + . " --genomeChrBinNbits $genomeChrBinNbits " + . " --genomeSAindexNbases $genomeSAindexNbases " + . " --limitGenomeGenerateRAM 40419136213 "; + + if ($gtf_file) { + + $cmd .= " --sjdbGTFfile $gtf_file " + . " --sjdbOverhang 150 "; + } + + $pipeliner->add_commands( new Command($cmd, $star_index_chkpt)); + + $pipeliner->run(); + + my $checkpoint_dir = "star_aln_chkpts." . basename($genome); + unless (-d $checkpoint_dir) { + mkdir($checkpoint_dir) or die "Error, cannot mkdir $checkpoint_dir"; + } + + + my $sort_opt = "SortedByCoordinate"; + my $sort_token = "c"; + my $bam_outfile = "Aligned.sortedByCoord.out.bam"; + if ($nameSorted) { + $sort_opt = "Unsorted"; + $sort_token = "n"; + $bam_outfile = "Aligned.out.bam"; + } + + + foreach my $sample_read_set (keys %sample_read_sets) { + + my $sample_id = $sample_read_set; + my $left_fqs = join(",", @{$sample_read_sets{$sample_id}->{left_fqs}}); + my $right_fqs = join(",", @{$sample_read_sets{$sample_id}->{right_fqs}}); + + + my $cmd = "$star_prog " + . " --runThreadN $CPU " + . " --genomeDir $star_index " + . " --outSAMtype BAM $sort_opt " + . " --runMode alignReads " + . " --readFilesIn $left_fqs $right_fqs " + . " --twopassMode Basic " + . " --alignSJDBoverhangMin 10 " + . " --outSAMstrandField intronMotif " + . " --outSAMunmapped Within " + . " --limitBAMsortRAM=20000000000" + . " --limitOutSJcollapsed=10000000" + . " --limitIObufferSize=150000000 300000000" + . " --limitSjdbInsertNsj=10000000 " + ; + + if (defined($max_intron_length)) { + $cmd .= " --alignMatesGapMax $max_intron_length " + . " --alignIntronMax $max_intron_length "; + } + + + if ($left_fqs =~ /\.gz$/) { + $cmd .= " --readFilesCommand 'gunzip -c' "; + } + + $pipeliner->add_commands( new Command($cmd, "$checkpoint_dir/star_align.$sample_id." . basename($genome) . ".ok") ); + + my $renamed_bam_outfile = "$sample_id.${sort_token}Sorted.star." . basename($genome) . ".bam"; + $pipeliner->add_commands( new Command("mv $bam_outfile $renamed_bam_outfile", "$checkpoint_dir/$renamed_bam_outfile.ok") ); + + unless ($nameSorted) { + $pipeliner->add_commands( new Command("samtools index $renamed_bam_outfile", "$checkpoint_dir/$renamed_bam_outfile.bai.ok") ); + } + + $pipeliner->add_commands( new Command("mv Log.final.out $sample_id.STAR.Log.final.out", "$checkpoint_dir/$sample_id.STAR_log_renamed.ok")); + + + $pipeliner->run(); + + + + } + + + exit(0); +} + + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + #return; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret ($ret)"; + } + + return; +} + +#### +sub parse_samples_file { + my ($samples_file) = @_; + + my %sample_to_fqs; + + open(my $fh, $samples_file) or die "Error, cannot open file $samples_file"; + while (<$fh>) { + unless (/\w/) { next; } + chomp; + my $line = $_; + my @x = split(/\t/); + my ($cond, $rep, $fq_a, $fq_b) = @x; + + unless ($fq_a) { + confess "Error, line in samples file: $samples_file, line: [$line] not formatted as expected (sample(tab)replicate(tab)left_fq(tab)right_fq)"; + } + + if (! defined $fq_b) { + $fq_b = ""; + } + + $fq_a = &Pipeliner::ensure_full_path($fq_a); + $fq_b = &Pipeliner::ensure_full_path($fq_b) if $fq_b; + + + if (! $join_bio_reps_flag) { + $cond = $rep; + } + + push (@{$sample_to_fqs{$cond}->{left_fqs}}, $fq_a); + push (@{$sample_to_fqs{$cond}->{right_fqs}}, $fq_b); + + } + close $fh; + + return (%sample_to_fqs); +} + diff --git a/99.scripts/trinity_utils/util/misc/run_Stringtie_via_bam_file_list.pl b/99.scripts/trinity_utils/util/misc/run_Stringtie_via_bam_file_list.pl new file mode 100644 index 0000000..99579a8 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_Stringtie_via_bam_file_list.pl @@ -0,0 +1,148 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib("$FindBin::RealBin/../../PerlLib"); +use Pipeliner; +use File::Basename; +use Cwd; +use List::Util qw(min); + +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + + +my $usage = <<__EOUSAGE__; + +###################################################################### +# +# Required: +# --bam_file_list bam_file_list (format: sample_id(tab)/path/to/file.bam) +# +# Optional +# --gtf annotations in gtf format +# --CPU number of threads (default: 2) +# --SS_lib_type +# +# +####################################################################### + + +__EOUSAGE__ + + ; + + +my $bam_file_list; + +my $CPU = 2; + +my $help_flag; +my $gtf_file; + +my $SS_lib_type; + +&GetOptions( 'h' => \$help_flag, + + 'bam_file_list=s' => \$bam_file_list, + 'CPU=i' => \$CPU, + 'gtf=s' => \$gtf_file, + + 'SS_lib_type=s' => \$SS_lib_type, + ); + + +unless ($bam_file_list) { + die $usage; +} + +if ($help_flag) { + die $usage; +} + +if (@ARGV) { + die "Error, cannot recognize opts: @ARGV"; +} + + + +main: { + + ## ensure all full paths + $gtf_file = &Pipeliner::ensure_full_path($gtf_file) if $gtf_file; + + my @entries; + open(my $fh, $bam_file_list) or die $!; + while(<$fh>) { + chomp; + my ($sample_id, $bam_filename) = split(/\t/); + unless ($sample_id && $bam_filename) { + die "Error, bam file list doesn't have expected tab-delimited formatting at $_"; + } + unless (-e $bam_filename) { + die "Error, cannot locate file: $bam_filename"; + } + + push (@entries, { sample_id => $sample_id, + bam => $bam_filename } ); + + } + + my $pipeliner = new Pipeliner(-verbose => 1); + + + my @indiv_stringtie_outputs; + + foreach my $entry (@entries) { + my $cmd = "stringtie " . $entry->{bam} . + " -l STRG." . $entry->{sample_id} . " "; + + if ($SS_lib_type) { + if ($SS_lib_type =~ /^R/i) { + $cmd .= " --rf "; + } + elsif ($SS_lib_type =~ /^F/i) { + $cmd .= " --fr "; + } + else { + die "Error, cannot determine strand-specificity type from $SS_lib_type"; + } + } + + $cmd .= " -o " . $entry->{sample_id} . ".strg.gtf"; + + push (@indiv_stringtie_outputs, $entry->{sample_id} . ".strg.gtf"); + + if ($gtf_file) { + $cmd .= " -G $gtf_file "; + } + my $checkpoint = $entry->{sample_id} . ".strg.ok"; + + $pipeliner->add_commands( new Command($cmd, $checkpoint)); + + } + + $pipeliner->run(); + + + # merge transcripts + if (scalar(@indiv_stringtie_outputs) > 1) { + my $cmd = "stringtie --merge -o stringtie.merged.gtf -g 1 "; + if ($gtf_file) { + $cmd .= " -G $gtf_file "; + } + $cmd .= join(" ", @indiv_stringtie_outputs); + + $pipeliner->add_commands( new Command($cmd, "stringtie_merge.ok")); + + $pipeliner->run(); + + } + + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/run_TOPHAT.pl b/99.scripts/trinity_utils/util/misc/run_TOPHAT.pl new file mode 100644 index 0000000..e85749c --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_TOPHAT.pl @@ -0,0 +1,103 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use File::Basename; +use Cwd; +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling); + + +my $usage = <<_EOUSAGE_; + +################################################################## +# +# Required: +# +# --target : transcript sequences (eg. 'Trinity.fasta') +# +# If paired reads: +# +# --left :left reads +# --right :right reads +# +# Or, if unpaired reads: +# +# --single :single reads +# +# +# --paired_fragment_length :size of a read pair insert (def=300) +# +# +############################################################################################################# + + + +_EOUSAGE_ + + ; + + + +my ($target_fasta, $left_file, $right_file, $single_file, $SS_lib_type, $paired_fragment_length); + +# defaults: +$paired_fragment_length = 300; + + +&GetOptions( + + ## general opts + "target_fasta=s" => \$target_fasta, + "left=s" => \$left_file, + "right=s" => \$right_file, + "single=s" => \$single_file, + + "SS_lib_type=s" => \$SS_lib_type, + + "paired_fragment_length=i" => \$paired_fragment_length, + ); + + +## Check options set: + +unless ( ($left_file && $right_file) || $single_file) { + die $usage; +} + + +main: { + + my $cmd = "ln -s $target_fasta TARGET.fa"; + &process_cmd($cmd) unless (-e "TARGET.fa"); + + $cmd = "bowtie-build TARGET.fa TARGET"; + &process_cmd($cmd) unless (-s "TARGET.1.ebwt"); + + if ($left_file && $right_file) { + $cmd = "tophat --bowtie1 -i 5 -r $paired_fragment_length TARGET $left_file $right_file"; + &process_cmd($cmd); + } + else { + $cmd = "tophat --bowtie1 -i 5 TARGET $single_file"; + &process_cmd($cmd); + } + + + exit(0); +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} diff --git a/99.scripts/trinity_utils/util/misc/run_bowtie2.pl b/99.scripts/trinity_utils/util/misc/run_bowtie2.pl new file mode 100644 index 0000000..a67df55 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_bowtie2.pl @@ -0,0 +1,88 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use Process_cmd; +use Getopt::Long qw(:config posix_default no_ignore_case bundling pass_through); +use Carp; + + +my $CPU = 2; + +my $usage = <<__EOUSAGE__; + +############################################################################ +# +# --target target for alignment +# +# --left read_1.fq +# +# optional: +# +# --right read_2.fq +# +# --CPU number of threads (default: $CPU) +# +# --max_hits default 10 + + usage: $0 --target target.seq --left reads_1.fq [--right reads_2.fq --CPU 8] + + and you can pipe it into samtools to make a bam file: + + | samtools view -@ 8 -Sb - | samtools sort -@ 8 -m 4G - -o bowtie2.coordSorted.bam + +############################################################################# + + + +__EOUSAGE__ + + ; + + +my $help_flag; +my $target_seq; +my $reads_1_fq; +my $reads_2_fq; +my $max_hits = 10; + +&GetOptions ( 'h' => \$help_flag, + 'target=s' => \$target_seq, + 'left=s' => \$reads_1_fq, + 'right=s' => \$reads_2_fq, + 'CPU=i' => \$CPU, + 'max_hits=i' => \$max_hits, + ); + + +if ($help_flag) { die $usage; } + +unless ($target_seq && $reads_1_fq) { die $usage; } + + + +main: { + + unless (-s "$target_seq.1.bt2") { + my $cmd = "bowtie2-build $target_seq $target_seq 1>&2 "; + &process_cmd($cmd); + } + + my $format = ($reads_1_fq =~ /\.fq|\.fastq/) ? "-q" : "-f"; + + my $bowtie2_cmd = "bowtie2 --threads $CPU --local --no-unal -x $target_seq $format -k $max_hits"; + if ($reads_2_fq) { + $bowtie2_cmd .= " -1 $reads_1_fq -2 $reads_2_fq "; + } + else { + $bowtie2_cmd .= " -U $reads_1_fq "; + } + + + &process_cmd($bowtie2_cmd); + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/misc/run_bwa.pl b/99.scripts/trinity_utils/util/misc/run_bwa.pl new file mode 100644 index 0000000..600517a --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_bwa.pl @@ -0,0 +1,147 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use Cwd; +use File::Basename; +use Carp; +use Data::Dumper; + +use Getopt::Long qw(:config no_ignore_case bundling); + +$ENV{LC_ALL} = 'C'; # critical for proper sorting using [system "sort -k1,1 ..."] within the perl script + +my $usage = <<_EOUSAGE_; + +################################################################################################################ +# +# --left and --right (if paired reads) +# or +# --single (if unpaired reads) +# +# Required inputs: +# +# --target multi-fasta file containing the target sequences (should be named {refName}.fa ) +# +# --out_prefix|o output prefix (default: bwa) +# +# +# ## General options +# +# Any options after '--' are passed onward to the alignments programs (except BLAT -which has certain options exposed above). +# +# To set the number of processors used by BWA use: +# -- -t 16 +# You could also set other options such as 'mismatch penalty' (via -- -M INT), etc. +# +#################################################################################################################### + + + +_EOUSAGE_ + + ; + + +my $help_flag; +my $target_db; +my $left_file; +my $right_file; +my $single_file; + +my $output_prefix = "bwa"; + + +unless (@ARGV) { + die $usage; +} + +&GetOptions ( 'h' => \$help_flag, + + ## required inputs + 'left=s' => \$left_file, + 'right=s' => \$right_file, + + 'single=s' => \$single_file, + + + "target=s" => \$target_db, + + 'output_prefix|o=s' => \$output_prefix, + + ); + + + + + + +if ($help_flag) { die $usage; } + +unless ($target_db && -s $target_db) { + die $usage . "Must specify target_db and it must exist at that location"; +} + + +unless ( ($single_file && -e $single_file) + || + ($left_file && -e $left_file + && $right_file && -e $right_file)) { + die $usage . "sorry, cannot find $left_file and $right_file"; +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + + if ($ret) { + confess "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + + + +main: { + + + my $cmd = "bwa index $target_db"; + &process_cmd($cmd) unless (-e "$target_db.bwt");; + + $cmd = "samtools faidx $target_db"; + &process_cmd($cmd) unless (-e "$target_db.fai"); + + + if ($left_file && $right_file) { + + $cmd = "bwa aln @ARGV $target_db $left_file > $left_file.sai"; + &process_cmd($cmd); + + $cmd = "bwa aln @ARGV $target_db $right_file > $right_file.sai"; + &process_cmd($cmd); + + $cmd = "bwa sampe $target_db $left_file.sai $right_file.sai $left_file $right_file | samtools view -bS -F 4 - | samtools sort -o $output_prefix.bam"; + &process_cmd($cmd); + + } + else { + + $cmd = "bwa aln @ARGV $target_db $single_file > $single_file.sai"; + &process_cmd($cmd); + + $cmd = "bwa samse $target_db $single_file.sai $single_file | samtools view -bS -F 4 - | samtools sort -o $output_prefix.bam"; + &process_cmd($cmd); + } + + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/misc/run_bwasw_trinity.pl b/99.scripts/trinity_utils/util/misc/run_bwasw_trinity.pl new file mode 100644 index 0000000..8721168 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_bwasw_trinity.pl @@ -0,0 +1,53 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use File::Basename; + +my $usage = "usage: $0 Trinity.fasta reads.{fa,fq}\n\n"; + +my $trinity_fasta = $ARGV[0] or die $usage; +my $reads_fasta = $ARGV[1] or die $usage; + + +main: { + + my $cmd = "fasta_file_header_stripper.pl < $trinity_fasta > $trinity_fasta.noheader"; + &process_cmd($cmd); + + $cmd = "bwa index -a is $trinity_fasta.noheader"; + &process_cmd($cmd); + + $cmd = "samtools faidx $trinity_fasta.noheader"; + &process_cmd($cmd); + + my $outfile_prefix = basename($reads_fasta); + $cmd = "bwa bwasw $trinity_fasta.noheader $reads_fasta > $outfile_prefix.sam"; + &process_cmd($cmd); + + $cmd = "samtools view -bt $trinity_fasta.noheader.fai $outfile_prefix.sam > $outfile_prefix.bam"; + &process_cmd($cmd); + + $cmd = "samtools sort $outfile_prefix.bam -o $outfile_prefix.bam "; + &process_cmd($cmd); + + $cmd = "samtools index $outfile_prefix.bam"; + &process_cmd($cmd); + + exit(0); +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} diff --git a/99.scripts/trinity_utils/util/misc/run_jellyfish.pl b/99.scripts/trinity_utils/util/misc/run_jellyfish.pl new file mode 100644 index 0000000..5370f91 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_jellyfish.pl @@ -0,0 +1,71 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Cwd; +use FindBin; + +my $usage = "\n\nusage: $0 reads.fa hash_size\n\n"; + +my $reads_file = $ARGV[0] or die $usage; +my $hash_size = $ARGV[1] or die $usage; + + +my $JELLYFISH_DIR = $FindBin::RealBin . "/../../trinity-plugins/jellyfish-1.1.3"; +my $CPU = 4; +my $min_kmer_cov = 1; + +unless ($reads_file =~ /^\//) { + $reads_file = cwd() . "/$reads_file"; +} + + +my $workdir = "H_" . ($hash_size/1e9) . "G"; +mkdir($workdir) or die "Error, cannot mkdir $workdir"; +chdir ($workdir) or die "Error, cannot cd to $workdir"; + + +my $jelly_kmer_fa_file = "jellyfish.kmers.fa"; + +# my $jelly_hash_size = int( ($max_memory - $read_file_size)/7); # decided upon by Rick Westerman + +my $cmd = "$JELLYFISH_DIR/bin/jellyfish count -t $CPU -m 25 -s $hash_size "; + +# $cmd .= " --both-strands "; + +$cmd .= " $reads_file"; + +&process_cmd($cmd); + +my @kmer_db_files; + +foreach my $file () { + my $cmd = "$JELLYFISH_DIR/bin/jellyfish dump -L $min_kmer_cov $file >> $file.kmer_fa"; + &process_cmd($cmd); + + $cmd = "cat $file.kmer_fa >> $jelly_kmer_fa_file"; + &process_cmd($cmd); + +} + +$cmd = "grep '>' $jelly_kmer_fa_file | wc -l | tee kmer_count.txt"; +&process_cmd($cmd); + + +exit(0); + + +#### +sub process_cmd { + my ($cmd) = @_; + + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + diff --git a/99.scripts/trinity_utils/util/misc/run_read_simulator_per_fasta_entry.pl b/99.scripts/trinity_utils/util/misc/run_read_simulator_per_fasta_entry.pl new file mode 100644 index 0000000..6771e33 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_read_simulator_per_fasta_entry.pl @@ -0,0 +1,74 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + + + +my $usage = "usage: $0 file.fasta [require_proper_pairs_flag] [include_volcano_spread]\n\n"; + +my $fasta_file = $ARGV[0] or die $usage; +my $require_proper_pairs_flag = $ARGV[1] or die $usage; +my $include_volcano_spread = $ARGV[2] or die $usage; + +main: { + + my $sim_out_dir = "sim_data"; + unless (-d $sim_out_dir) { + mkdir $sim_out_dir or die $!; + } + + my $fasta_reader = new Fasta_reader($fasta_file); + + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + my $sequence = $seq_obj->get_sequence(); + + my $outdir = $acc; + $outdir =~ s/\W/_/g; + + + mkdir ("$sim_out_dir/$outdir") or die $!; + + my $template_file = "$sim_out_dir/$outdir/$outdir.template.fa"; + open (my $ofh, ">$template_file") or die "Error, cannot write to $template_file"; + print $ofh ">$acc\n$sequence\n"; + close $ofh; + + my $outfile = "$sim_out_dir/$outdir/$outdir.reads.fa"; + + my $cmd = "$FindBin::RealBin/simulate_illuminaPE_from_transcripts.pl --transcripts $template_file --out_prefix $template_file"; + if ($require_proper_pairs_flag) { + $cmd .= " --require_proper_pairs"; + } + if ($include_volcano_spread) { + $cmd .= " --include_volcano_spread"; + } + + &process_cmd($cmd); + + } + + exit(0); + +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + diff --git a/99.scripts/trinity_utils/util/misc/run_read_simulator_per_gene.pl b/99.scripts/trinity_utils/util/misc/run_read_simulator_per_gene.pl new file mode 100644 index 0000000..4f1aee4 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_read_simulator_per_gene.pl @@ -0,0 +1,102 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use lib ($ENV{EUK_MODULES}); +use Fasta_reader; +use FindBin; + + +my $usage = "usage: $0 file.fasta [max_genes]\n\n"; + +my $fasta_file = $ARGV[0] or die $usage; +my $max_genes = $ARGV[1]; + + +main: { + + my $sim_out_dir = "sim_AS_data"; + unless (-d $sim_out_dir) { + mkdir $sim_out_dir or die $!; + } + + my $fasta_reader = new Fasta_reader($fasta_file); + + my %gene_to_seqs; + + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + + my ($trans, $gene) = split(/;/, $acc); + + unless ($gene) { + die "Error, need trans;gene format for accession: $acc"; + } + + my $sequence = $seq_obj->get_sequence(); + + push (@{$gene_to_seqs{$gene}}, { acc => $acc, + seq => $sequence, }); + + } + + + my $gene_counter = 0; + ## only including those entries that are alt-spliced + foreach my $gene (keys %gene_to_seqs) { + + my @trans = @{$gene_to_seqs{$gene}}; + + if (scalar @trans == 1) { + next; + } + + + my $outdir = $gene; + $outdir =~ s/\W/_/g; + + mkdir ("$sim_out_dir/$outdir") or die $!; + + my $template_file = "$sim_out_dir/$outdir/$outdir.template.fa"; + open (my $ofh, ">$template_file") or die "Error, cannot write to $template_file"; + foreach my $entry (@trans) { + my ($acc, $sequence) = ($entry->{acc}, $entry->{seq}); + print $ofh ">$acc\n$sequence\n"; + } + close $ofh; + + my $outfile = "$sim_out_dir/$outdir/$outdir.reads.fa"; + + my $cmd = "$FindBin::RealBin/simulate_illuminaPE_from_transcripts.pl --transcripts $template_file --SS --out_prefix $sim_out_dir/$outdir/reads"; + &process_cmd($cmd); + + $gene_counter++; + + if ($max_genes && $gene_counter >= $max_genes) { + last; + } + + + } + + exit(0); + +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + diff --git a/99.scripts/trinity_utils/util/misc/run_trimmomatic_qual_trimming.pl b/99.scripts/trinity_utils/util/misc/run_trimmomatic_qual_trimming.pl new file mode 100644 index 0000000..f0784c3 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/run_trimmomatic_qual_trimming.pl @@ -0,0 +1,109 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + +use FindBin; + +my $usage = <<__EOUSAGE__; + +############################################################### +# +# --left left.fq +# --right right.fq +# +# or +# +# --single single.fq +# +# Optional: +# +# --CPU default: 4 +# +# --trim_params "SLIDINGWINDOW:4:5 LEADING:5 TRAILING:5 MINLEN:25" +# +############################################################### + +__EOUSAGE__ + + + ; + + +my $left; +my $right; +my $single; + +my $threads = 4; +my $trim_params = "SLIDINGWINDOW:4:5 LEADING:5 TRAILING:5 MINLEN:25"; + +&GetOptions( 'left=s' => \$left, + 'right=s' => \$right, + 'single=s' => \$single, + + 'CPU=i' => \$threads, + + 'trim_params=s' => \$trim_params, + + ); + + +=trimmomatic + +java -jar /seq/regev_genome_portal/SOFTWARE/BIN/trimmomatic.jar PE -threads {__THREADS__} -phred33 \ +{__LEFT_FQ__} {__RIGHT_FQ__} \ +{__LEFT_FQ__}.P.qtrim.fq {__LEFT_FQ__}.U.qtrim.fq \ +{__RIGHT_FQ__}.P.qtrim.fq {__RIGHT_FQ__}.U.qtrim.fq \ + LEADING:15 TRAILING:15 MINLEN:36 2> {__LOCAL_ANALYSIS_DIR__}/trimmomatic.log.stats + +=cut + + ; + +unless ( ($left && $right) || $single) { + die $usage; +} + +main: { + + my $cmd; + + if ($left && $right) { + + $cmd = "java -jar $FindBin::RealBin/../../trinity-plugins/Trimmomatic/trimmomatic.jar PE -threads $threads -phred33 " + . " $left $right " + . " $left.P.qtrim.fq $left.U.qtrim.fq " + . " $right.P.qtrim.fq $right.U.qtrim.fq " + . " $trim_params "; + } + else { + + $cmd = "java -jar $FindBin::RealBin/../../trinity-plugins/Trimmomatic/trimmomatic.jar SE -threads $threads -phred33 " + . " $single " + . " $single.qtrim.fq " + . " $trim_params "; + + } + + &process_cmd($cmd); + + exit(0); +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; + +} diff --git a/99.scripts/trinity_utils/util/misc/seqinfo_refseq_to_dot.pl b/99.scripts/trinity_utils/util/misc/seqinfo_refseq_to_dot.pl new file mode 100644 index 0000000..7c2b6f7 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/seqinfo_refseq_to_dot.pl @@ -0,0 +1,154 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib", "$FindBin::RealBin/../../PerlLib/KmerGraphLib"); +use Fasta_reader; +use ColorGradient; +use Data::Dumper; + + +my $usage = "usage: $0 graph.seqinfo refseqs.fa\n\n"; + +my $graph_seqinfo_file = $ARGV[0] or die $usage; +my $refseqs_fa_file = $ARGV[1] or die $usage; + +main: { + + my $fasta_reader = new Fasta_reader($refseqs_fa_file); + my %refseq_fa = $fasta_reader->retrieve_all_seqs_hash(); + + my ($node_seq_to_node_id_href, $edges_aref) = &parse_seqinfo($graph_seqinfo_file); + + my %refseq_to_nodes_list; + foreach my $refseq_acc (keys %refseq_fa) { + my $refseq_seq = $refseq_fa{$refseq_acc}; + + foreach my $node_seq (keys %$node_seq_to_node_id_href) { + my $node_id_info = $node_seq_to_node_id_href->{$node_seq}; + my $idx = &get_node_start_pos($refseq_seq, $node_seq); + + if ($idx >= 0) { + + my ($node_id, @rest) = split(/\s+/, $node_id_info); + + push (@{$refseq_to_nodes_list{$refseq_acc}}, { start_pos => $idx, + node_id => $node_id, + node_info => $node_id_info, + + } ); + } + } + } + + my $dot_file = "$graph_seqinfo_file.dot"; + open(my $ofh, ">$dot_file") or die "Error, cannot write to file: $dot_file"; + + # print Dumper(\%refseq_to_nodes_list); + + ## output graph: + print $ofh "digraph G {\n" + . " node [width=0.1,height=0.1,fontsize=10];\n" + . " edge [fontsize=12];\n" + . " margin=1.0;\n" + . " rankdir=LR;\n" + . " labeljust=l;\n"; + + + + + foreach my $node_seq (keys %$node_seq_to_node_id_href) { + my $node_info_txt = $node_seq_to_node_id_href->{$node_seq}; + print $ofh " $node_info_txt\n"; + } + + ## get a different color for each refseq acc: + my @colors = &ColorGradient::convert_RGB_hex(&ColorGradient::get_RGB_gradient(scalar keys %refseq_fa)); + + foreach my $acc (keys %refseq_to_nodes_list) { + my @structs = @{$refseq_to_nodes_list{$acc}}; + @structs = sort {$a->{start_pos}<=>$b->{start_pos}} @structs; + + my $color = shift @colors; + + ## examine each edge + for (my $i = 0; $i < $#structs; $i++) { + my $node_id_begin = $structs[$i]->{node_id}; + my $node_id_end = $structs[$i+1]->{node_id}; + + print $ofh " $node_id_begin->$node_id_end [label=\"$acc\", color=\"$color\"];\n"; + } + } + + foreach my $edge (@$edges_aref) { + print $ofh " $edge;\n"; + } + + print $ofh "}\n"; + + close $ofh; + + + exit(system("dot -Tpdf $dot_file > $dot_file.pdf")); + + + +} + +#### +sub parse_seqinfo { + my ($seqinfo_file) = @_; + + my %node_seq_to_node_id; + my @edges; + + open(my $fh, $seqinfo_file) or die "Error, cannot open file: $seqinfo_file"; + while(<$fh>) { + chomp; + my @x = split(/\t/); + if (scalar(@x) == 1) { + my $edge = $x[0]; + if ($edge =~ /(\d+)->(\d+)/) { + push(@edges, $edge); + } + else { + die "Error, cannot identify $edge as an edge"; + } + } + else { + my ($node_descr, $node_seq) = @x; + $node_seq_to_node_id{$node_seq} = $node_descr; + } + } + + close $fh; + + return(\%node_seq_to_node_id, \@edges); +} + +#### +sub get_node_start_pos { + my ($refseq_seq, $node_seq) = @_; + + my $idx = index($refseq_seq, $node_seq); + if ($idx >= 0) { + return($idx); + } + else { + ## try 1st kmer + my $first_kmer = substr($node_seq, 0, 25); + $idx = index($refseq_seq, $first_kmer); + if ($idx) { + return($idx); + } + else { + # try last kmer + my $last_kmer = substr($node_seq, -25); + $idx = index($refseq_seq, $last_kmer); + return($idx); + } + } +} + + diff --git a/99.scripts/trinity_utils/util/misc/shuffle.pl b/99.scripts/trinity_utils/util/misc/shuffle.pl new file mode 100644 index 0000000..7c0e52c --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/shuffle.pl @@ -0,0 +1,22 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use List::Util qw (shuffle); + +my @list; + +while () { + push (@list, $_); +} + +@list = shuffle @list; + + +foreach my $ele (@list) { + print $ele; +} + + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/sim_test_framework/audit_summary_stats.pl b/99.scripts/trinity_utils/util/misc/sim_test_framework/audit_summary_stats.pl new file mode 100644 index 0000000..8c33a13 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/sim_test_framework/audit_summary_stats.pl @@ -0,0 +1,46 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 file.audit_summary\n\n"; + +my $filename = $ARGV[0] or die $usage; + +main: { + + my %yes_no_counter; + my $total_reco = 0; + my $total_ref = 0; + my $total_FL = 0; + + open (my $fh, $filename) or die $!; + while (<$fh>) { + if (/ref_fa/) { next; } + chomp; + my @x = split(/\t/); + my $yes_or_no = $x[4]; + + $yes_no_counter{$yes_or_no}++; + + my $num_reco = $x[3]; + $total_reco += $num_reco; + + my $num_ref = $x[1]; + $total_ref += $num_ref; + + my $num_FL = $x[2]; + $total_FL += $num_FL; + + } + close $fh; + + my $num_yes = $yes_no_counter{YES} || 0; + my $num_no = $yes_no_counter{NO} ||0; + + my $num_additional = $total_reco - $total_FL; + print join("\t", "#YES", "#NO", "#FL", "#TOT_Trans", "#Ref", "#additional") . "\n"; + print join("\t", $num_yes, $num_no, $total_FL, $total_reco, $total_ref, $num_additional) . "\n"; + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/misc/sim_test_framework/audit_summary_stats.reexamine.pl b/99.scripts/trinity_utils/util/misc/sim_test_framework/audit_summary_stats.reexamine.pl new file mode 100644 index 0000000..a397d00 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/sim_test_framework/audit_summary_stats.reexamine.pl @@ -0,0 +1,159 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use File::Basename; + +use lib ($ENV{EUK_MODULES}); +use Fasta_reader; + +my $usage = "\n\n\tusage: $0 < list of audit.txt files from stdin \n\n\n"; + +my $MIN_SEQ_LEN = 1000; + + + +my $total_genes = 0; +my $total_genes_reco = 0; + +my $total_refseq_trans = 0; +my $total_iso_reco_count = 0; +my $total_extra = 0; + +print join("\t", "gene_id", "reco_gene_flag", "num_refseqs", "num_FL_trin", "num_extra_trin") . "\n"; + +my $counter = 0; +while (<>) { + chomp; + my $dir = dirname($_); + + $counter++; + + my $gene_id = basename($dir); + + my $trin_fasta_file = "$dir/trinity_out_dir.Trinity.fasta"; + + if (! -e $trin_fasta_file) { + print STDERR "ERROR: Cannot locate file: $trin_fasta_file\n"; + next; + } + + my $fasta_reader = new Fasta_reader($trin_fasta_file); + my %trin_seqs = $fasta_reader->retrieve_all_seqs_hash(); + + my %trin_lens = &get_seq_lens(%trin_seqs); + + my $FL_reco_file = "$dir/FL.test.pslx.maps"; + my %reco_refseq; + my %reco_trin_to_refseq = &parse_reco($FL_reco_file, \%reco_refseq); + my $num_FL_trin = scalar(keys %reco_trin_to_refseq); + + my $refseqs_fa = "$dir/refseqs.fa"; + my @refseq_accs = &get_accs($refseqs_fa); + + my $num_refseqs = scalar(@refseq_accs); + + my @failed_reco_refseqs; + + foreach my $acc (@refseq_accs) { + unless ($reco_refseq{$acc}) { + push (@failed_reco_refseqs, $acc); + } + } + my $num_failed_reco_refseqs = scalar(@failed_reco_refseqs); + + my @extra_trin_accs; + foreach my $trin_acc (keys %trin_seqs) { + if (! exists $reco_trin_to_refseq{$trin_acc}) { + if ($trin_lens{$trin_acc} >= $MIN_SEQ_LEN) { + push (@extra_trin_accs, $trin_acc); + } + } + } + + my $num_extra_trin = scalar(@extra_trin_accs); + + my $reco_gene_flag = ($num_failed_reco_refseqs == 0) ? "YES" : "NO"; + + print join("\t", $gene_id, $reco_gene_flag, $num_refseqs, $num_FL_trin, $num_extra_trin) . "\n"; + + $total_genes++; + if ($reco_gene_flag eq "YES") { + $total_genes_reco++; + } + $total_refseq_trans += $num_refseqs; + $total_iso_reco_count += $num_FL_trin; + $total_extra += $num_extra_trin; + +} + +if ($counter > 1) { + print "\n\n"; + print join("\t", "Total_Genes", "Total_Genes_Reco", "Total_RefTrans", "Total_RefTransReco", "Total_extra_trans") . "\n"; + print join("\t", $total_genes, $total_genes_reco, $total_refseq_trans, $total_iso_reco_count, $total_extra) . "\n"; +} + + +exit(0); + + +#### +sub get_accs { + my ($fasta_file) = @_; + + my @accs; + + open (my $fh, $fasta_file) or die $!; + while (<$fh>) { + if (/^>(\S+)/) { + push (@accs, $1); + } + } + + return(@accs); +} + + +#### +sub parse_reco { + my ($reco_file, $reco_refseq_href) = @_; + + my %trin_to_reco_acc; + + open (my $fh, $reco_file) or die $!; + while (<$fh>) { + chomp; + my ($trans_acc, $trinity_contigs) = split(/\t/); + + my @trin_contigs = split(/,/, $trinity_contigs); + foreach my $trin (@trin_contigs) { + + $trin_to_reco_acc{$trin}->{$trans_acc} = 1; + + $reco_refseq_href->{$trans_acc} = 1; + } + } + close $fh; + + return(%trin_to_reco_acc); + +} + + +#### +sub get_seq_lens { + my (%trin_seqs) = @_; + + my %lens; + + foreach my $acc (keys %trin_seqs) { + my $seq = $trin_seqs{$acc}; + my $seqlen = length($seq); + + $lens{$acc} = $seqlen; + } + + return(%lens); + +} + diff --git a/99.scripts/trinity_utils/util/misc/sim_test_framework/info_files_to_eval_cmds.pl b/99.scripts/trinity_utils/util/misc/sim_test_framework/info_files_to_eval_cmds.pl new file mode 100644 index 0000000..260d072 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/sim_test_framework/info_files_to_eval_cmds.pl @@ -0,0 +1,46 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Cwd; +use FindBin; + +my $usage = "usage: $0 info_files.list.txt output_basedir [eval cmds]\n\n"; + +my $files_listing_file = $ARGV[0] or die $usage; +my $output_basedir = $ARGV[1] or die $usage; +shift @ARGV; +shift @ARGV; + + +main: { + + + my @files = `cat $files_listing_file`; + chomp @files; + + my $eval_script = "$FindBin::Bin/run_Trinity_eval.sh"; + my $basedir = cwd(); + + unless ($output_basedir =~ /^\//) { + $output_basedir = "$basedir/$output_basedir"; + } + + foreach my $file (@files) { + + my $line = `cat $file`; + chomp $line; + my ($refseq_fa_file, $left_fa, $right_fa) = split(/\t/, $line); + + my @pts = split(/\//, $refseq_fa_file); + my $gene_name = $pts[-2]; + + my $cmd = "$eval_script -R $refseq_fa_file --left $left_fa --right $right_fa -O $output_basedir/$gene_name @ARGV"; + + print "$cmd\n"; + } + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/sim_test_framework/partition_target_transcripts.pl b/99.scripts/trinity_utils/util/misc/sim_test_framework/partition_target_transcripts.pl new file mode 100644 index 0000000..4a6df12 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/sim_test_framework/partition_target_transcripts.pl @@ -0,0 +1,308 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$ENV{TRINITY_HOME}/PerlLib/"); +use Fasta_reader; +use Cwd; +use Data::Dumper; +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use List::Util qw (shuffle); + +my $help_flag; + +my $ref_trans_fa; +my $MIN_REFSEQ_LENGTH = 100; +my $OUT_DIR = "Seqs_dir"; + +my $MAX_ISOFORMS = -1; +my $MIN_ISOFORMS = 2; + + +my $usage = <<__EOUSAGE__; + +################################################################################ +# +# * Required: +# +# --ref_trans|R reference transcriptome +# +# * Common Opts: +# +# --by_Gene target all isoforms of a gene at once. +# (requires multiple isoforms, ignores single-iso genes) +# +# --out_dir|O output directory name (default: $OUT_DIR) +# +# * Misc Opts: +# +# --min_refseq_length min length for a reference transcript +# sequence (default: $MIN_REFSEQ_LENGTH) +# +# if --by_Gene: +# +# --min_isoforms default: $MIN_ISOFORMS +# --max_isoforms max number of isoforms to test (default: $MAX_ISOFORMS) +# +# --longest_isoform_only restricts to only single longest isoform per gene. +# +# --restrict_to_genes file containing lists of gene accessions to restrict to. +# +############################################################################################ + + + +__EOUSAGE__ + + ; + + + +my $BY_GENE_FLAG = 0; +my $LONGEST_ISOFORM_ONLY_FLAG = 0; + +my $restrict_to_genes_file = ""; + +&GetOptions ( 'h' => \$help_flag, + + # required + 'ref_trans|R=s' => \$ref_trans_fa, + + # optional + 'out_dir|O=s' => \$OUT_DIR, + 'min_refseq_length=i' => \$MIN_REFSEQ_LENGTH, + + 'by_Gene' => \$BY_GENE_FLAG, + + 'max_isoforms=i' => \$MAX_ISOFORMS, + 'min_isoforms=i' => \$MIN_ISOFORMS, + + 'longest_isoform_only' => \$LONGEST_ISOFORM_ONLY_FLAG, + + 'restrict_to_genes=s' => \$restrict_to_genes_file, +); + + + +if ($help_flag) { + die $usage; +} + +unless ($ref_trans_fa) { + die $usage; +} + + +main: { + + my $BASEDIR = cwd(); + + + if ($ref_trans_fa =~ /\.gz$/) { + my $unzipped = $ref_trans_fa; + $unzipped =~ s/\.gz$//g; + if (! -s $unzipped) { + &process_cmd("gunzip -c $ref_trans_fa > $unzipped"); + } + + $ref_trans_fa = $unzipped; + } + + my $fasta_reader = new Fasta_reader($ref_trans_fa); + my %fasta_seqs = $fasta_reader->retrieve_all_seqs_hash(); + + my %reorganized_fasta_seqs = &reorganize_fasta_seqs(\%fasta_seqs, $BY_GENE_FLAG); + + + my %restricted_genes; + if ($restrict_to_genes_file) { + my @gene_ids = `cat $restrict_to_genes_file`; + chomp @gene_ids; + %restricted_genes = map { + $_ => 1 } @gene_ids; + } + + my $total_counter = 0; + + my @accs = keys %reorganized_fasta_seqs; + + my %seen; + + foreach my $acc (@accs) { + + if (%restricted_genes && ! exists $restricted_genes{$acc}) { + # skipping, not in the restricted list. + next; + } + $seen{$acc} = 1; + + chdir $BASEDIR or die "Error, cannot cd to $BASEDIR"; + + + my $seq_entries_aref = $reorganized_fasta_seqs{$acc}; + + my @min_length_targets; + foreach my $entry (@$seq_entries_aref) { + + my ($trans_acc, $seq) = ($entry->{acc}, + $entry->{seq}); + + if (length($seq) >= $MIN_REFSEQ_LENGTH && $seq !~ /[^GATC]/i) { + push (@min_length_targets, $entry); + } + } + + unless (@min_length_targets) { + print STDERR "No min length targets to pursue for $acc .... skipping.\n"; + next; + } + + unless (-d $OUT_DIR) { + mkdir($OUT_DIR) or die $!; + } + + @min_length_targets = reverse sort {length($a->{seq}) <=> length($b->{seq}) } @min_length_targets; + + my $num_total_targets = scalar(@min_length_targets); + + if ($BY_GENE_FLAG) { + + if ($LONGEST_ISOFORM_ONLY_FLAG) { + @min_length_targets = shift @min_length_targets; + } + else { + + if ($num_total_targets < $MIN_ISOFORMS) { + next; + } + + if ($num_total_targets > $MAX_ISOFORMS) { + + @min_length_targets = @min_length_targets[0..($num_total_targets-1)]; + } + } + + } + + &prep_seqs($acc, \@min_length_targets); + + $total_counter++; + if ($total_counter % 100 == 0) { + print STDERR "\n[$total_counter]\n"; + } + } + + + if (%restricted_genes) { + # ensure we got them all + for my $seen_acc (keys %seen) { + if (exists $restricted_genes{$seen_acc}) { + delete $restricted_genes{$seen_acc}; + } + } + + if (%restricted_genes) { + die "Error, missing entries for restricted gene list entries: " . Dumper(\%restricted_genes); + } + else { + print STDERR "-all restricted gene entries identified and reported.\n"; + } + } + + print STDERR "\nDone.\n\n"; + + exit(0); +} + + + +#### +sub prep_seqs { + my ($acc, $entries_aref) = @_; + + print STDERR "\r-processing $acc "; + + my $num_entries = scalar(@$entries_aref); + + my $basedir = cwd(); + + my $dir_tok = $acc; + $dir_tok =~ s/\W/_/g; + + my $workdir = "$OUT_DIR/$dir_tok"; + + unless (-d $workdir) { + mkdir $workdir or die "Error, cannot mkdir $workdir"; + } + + my $refseqs_fa = "$workdir/refseqs.fa"; + + # write ref fasta seqs. + if (! -s "$refseqs_fa") { + open (my $ofh, ">$refseqs_fa") or die $!; + foreach my $entry (@$entries_aref) { + my ($entry_acc, $seq) = ($entry->{acc}, + $entry->{seq}); + + print $ofh ">$entry_acc\n$seq\n"; + } + close $ofh; + } + + return; + +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + +#### +sub reorganize_fasta_seqs { + my ($fasta_seqs_href, $by_gene_flag) = @_; + + my %reorg_fasta; + + foreach my $acc (sort keys %$fasta_seqs_href) { + + my $seq = uc $fasta_seqs_href->{$acc}; + + my $key = $acc; + if ($by_gene_flag) { + if ($acc =~ /^([^;]+);([^;]+)$/) { + my $trans = $1; + my $gene = $2; + $key = $gene; + } + elsif ($acc =~ /^([^\|]+)\|([^\|]+)$/) { + my $gene = $1; + my $trans = $2; + $key = $gene; + } + else { + confess "Error, no gene ID extracted from $acc "; + } + + } + + push (@{$reorg_fasta{$key}}, { acc => $acc, + seq => $seq} + ); + } + + return(%reorg_fasta); + +} + diff --git a/99.scripts/trinity_utils/util/misc/sim_test_framework/run_Trinity_eval.pl b/99.scripts/trinity_utils/util/misc/sim_test_framework/run_Trinity_eval.pl new file mode 100644 index 0000000..99fb157 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/sim_test_framework/run_Trinity_eval.pl @@ -0,0 +1,461 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$ENV{TRINITY_HOME}/PerlLib/"); +use Fasta_reader; +use Cwd; +use Data::Dumper; +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use List::Util qw (shuffle); + +my $VERBOSITY_LEVEL = 10; + + +my $help_flag; +my $ref_trans_fa; +my $BFLY_JAR = "$ENV{TRINITY_HOME}/Butterfly/Butterfly.jar"; +my $INCLUDE_REF_TRANS = 0; +my $OUT_DIR = "testing_dir"; +my $MIN_CONTIG_LENGTH = 200; +my $min_per_id = 90; + + +my $usage = <<__EOUSAGE__; + +################################################################################ +# +# * Required: +# +# --ref_trans|R reference transcriptome +# +# --left left reads fa file +# +# --right right reads fa file +# +# * Common Opts: +# +# --bfly_jar|B Butterfly jar file +# +# --out_dir|O output directory name (default: $OUT_DIR) +# +# * include FL seq opts: +# +# --incl_ref_trans include the ref transcript as a long +# read (default: off) +# +# * Misc Opts: +# +# --incl_ref_dot include dot files for the reference sequences +# +# -V verbosity level (default: 12) +# +# --acc restrict to a specific accession (gene or transcript) +# +# --paired_as_single treat paired reads as single reads +# +# --min_contig_length minimum contig length for Trinity assembly. default: $MIN_CONTIG_LENGTH +# +# --strict weld all, no pruning or path merging. +# +# --no_cleanup no cleaning up of trinity output. +# +# --strand_specific sets to strand-specific mode (RF) +# +# --bfly_opts butterfly additional opts +# +############################################################################################ + + +__EOUSAGE__ + + ; + + +my $NO_CLEANUP = 0; + +my $INCLUDE_REF_DOT_FILES = 0; + +my $PAIRED_AS_SINGLE = ""; + +my $SHUFFLE = 0; + +my $MAX_ISOFORMS = -1; + +my $STRICT = 0; + + +my $strand_specific_flag = 0; + +my $left_file = ""; +my $right_file = ""; +my $BFLY_OPTS; + +&GetOptions ( 'h' => \$help_flag, + + # required + 'ref_trans|R=s' => \$ref_trans_fa, + + 'left=s' => \$left_file, + 'right=s' => \$right_file, + + # optional + 'out_dir|O=s' => \$OUT_DIR, + 'bfly_jar|B=s' => \$BFLY_JAR, + + 'incl_ref_trans' => \$INCLUDE_REF_TRANS, + + 'strict' => \$STRICT, + + 'V=i' => \$VERBOSITY_LEVEL, + + 'incl_ref_dot' => \$INCLUDE_REF_DOT_FILES, + + 'paired_as_single' => \$PAIRED_AS_SINGLE, + + 'min_contig_length=i' => \$MIN_CONTIG_LENGTH, + + 'no_cleanup' => \$NO_CLEANUP, + + 'strand_specific' => \$strand_specific_flag, + + 'min_per_id=i' => \$min_per_id, + + 'bfly_opts=s' => \$BFLY_OPTS, +); + + +if ($help_flag) { + die $usage; +} + + + +unless ($ref_trans_fa && $BFLY_JAR && $left_file && $right_file) { + die $usage; +} + + +$NO_CLEANUP = 1; ## NEEDED NOW for iworm and bfy pruning assessment + +unless ($ENV{TRINITY_HOME}) { + $ENV{TRINITY_HOME} = "$FindBin::Bin/../../trinityrnaseq/"; +} + +my $reconstructions_log_file = "$OUT_DIR.reconstruction_summary.txt"; + + +if ($PAIRED_AS_SINGLE) { + $PAIRED_AS_SINGLE = "--TREAT_PAIRS_AS_SINGLE"; +} + + + + +main: { + + my $BASEDIR = cwd(); + + unless ($ref_trans_fa =~ /^\//) { + $ref_trans_fa = "$BASEDIR/$ref_trans_fa"; + } + + + unless (-d $OUT_DIR) { + &process_cmd("mkdir -p $OUT_DIR"); + } + chdir $OUT_DIR or die "Error, cannot cd to $OUT_DIR"; + + + + my ($num_reco_FL, $num_ref_entries, $num_trans_reco, + $has_all_iworm_kmers, $has_all_precious_edges, + $num_LR_threaded) = &execute_seq_pipe($ref_trans_fa, $left_file, $right_file); + # num_FL: number of transcripts reconstructed as full-length + # num_entries: number of reference isoforms + # num_transcripts: total number of Trinity transcripts reconstructed. + + open (my $ofh, ">audit.txt") or die $!; + my $captured_all = ($num_reco_FL == $num_ref_entries) ? "YES" : "NO"; + + + my $header = join("\t", "ref_fa", "num_ref", "num_FL", "num_reco", "captured_all", "iworm_ok", "bfly_edges_ok", + "num_LR_threaded", "all_LR_threaded_ok"); + + my $all_LR_threaded_ok = ($num_LR_threaded == $num_ref_entries) ? "YES" : "NO"; + + my $summary = join("\t", $ref_trans_fa, + $num_ref_entries, + $num_reco_FL, + $num_trans_reco, + $captured_all, + $has_all_iworm_kmers, + $has_all_precious_edges, + $num_LR_threaded, + $all_LR_threaded_ok); + + $summary = "$header\n$summary\n"; + + print $ofh $summary; + print $summary; + close $ofh; + + &process_cmd("echo " . cwd() . "/audit.txt | $FindBin::Bin/audit_summary_stats.reexamine.pl | tee audit2.txt"); + + + exit(0); +} + + + +#### +sub execute_seq_pipe { + my ($ref_trans_fa, $left_file, $right_file) = @_; + + + my $cmd = "ln -sf $ref_trans_fa $left_file $right_file ."; + &process_cmd($cmd); + + if (-d "trinity_out_dir") { + `rm -rf ./trinity_out_dir`; + } + + my $num_entries = `grep '>' $ref_trans_fa | wc -l `; + $num_entries =~ /(\d+)/ or die "Error, cannot parse number of entries from $ref_trans_fa"; + $num_entries = $1; + + # run Trinity + + my $bfly_jar_txt = ""; + if ($BFLY_JAR) { + $bfly_jar_txt = " --bfly_jar $BFLY_JAR "; + } + + $cmd = "set -o pipefail; $ENV{TRINITY_HOME}/Trinity --seqType fa --max_memory 4G --bflyHeapSpaceMax 10G --max_reads_per_graph 10000000 --group_pairs_distance 10000 --verbose_level 2 --CPU 1 "; + if ($NO_CLEANUP) { + $cmd .= " --no_cleanup "; + } + else { + $cmd .= " --full_cleanup "; + } + + if ($INCLUDE_REF_TRANS) { + $cmd .= " --long_reads $ref_trans_fa "; + } + + $cmd .= " --left $left_file --right $right_file "; + + if ($strand_specific_flag) { + $cmd .= " --SS_lib_type RF "; + } + + + $cmd .= " --CPU 2 $bfly_jar_txt --inchworm_cpu 1 --min_contig_length $MIN_CONTIG_LENGTH --trinity_complete @ARGV"; + + + + my $bfly_opts = " --bfly_opts \"--generate_intermediate_dot_files -R 1 --generate_intermediate_dot_files $PAIRED_AS_SINGLE --stderr -V $VERBOSITY_LEVEL $BFLY_OPTS\" "; + + if ($STRICT) { + $cmd .= " --no_bowtie --chrysalis_debug_weld_all " + . " --iworm_opts \"--no_prune_error_kmers --min_assembly_coverage 1 --min_seed_entropy 0 --min_seed_coverage 1 \" "; + + $bfly_opts = " --bfly_opts \"--dont-collapse-snps --no_pruning --no_path_merging --no_remove_lower_ranked_paths --NO_EM_REDUCE --MAX_READ_SEQ_DIVERGENCE=0 --NO_DP_READ_TO_VERTEX_ALIGN --generate_intermediate_dot_files -R 1 -F 100000 --generate_intermediate_dot_files $PAIRED_AS_SINGLE --stderr -V $VERBOSITY_LEVEL $BFLY_OPTS\" "; + + } + else { + #$cmd .= " --no_bowtie --chrysalis_debug_weld_all "; + #$cmd .= " --iworm_opts \" --min_seed_entropy 1 \" --min_glue 1 "; + } + + $cmd .= " $bfly_opts 2>&1 | tee trin.log"; + + + { + open (my $ofh, ">runTrinity.cmd") or die $!; + print $ofh $cmd; + close $ofh; + } + + &process_cmd($cmd); + + if ($NO_CLEANUP) { + rename("trinity_out_dir/Trinity.fasta", "trinity_out_dir.Trinity.fasta"); + } + + + ## check inchworm kmer content of reference sequences + + &process_cmd("$ENV{TRINITY_HOME}/util/misc/print_kmers.pl $ref_trans_fa 24 > ref_kmers"); + + + my $iworm_file = (-s "trinity_out_dir/inchworm.K25.L25.fa") ? "trinity_out_dir/inchworm.K25.L25.fa" : "trinity_out_dir/inchworm.K25.L25.DS.fa"; + my $has_all_iworm_kmers = &check_inchworm_kmer_content($ref_trans_fa, $iworm_file); + + + ## examine pruning of precious edges + my ($has_all_precious_edges) = &check_pruning("trin.log", "ref_kmers"); + + ## see if all LR are threaded through the graph + my $num_LR_threaded = &count_num_LR_threaded("trin.log"); + + if ($INCLUDE_REF_DOT_FILES) { + # generate sequence graphs just refseqs + $cmd = "$ENV{TRINITY_HOME}/util/misc/Monarch --misc_seqs $ref_trans_fa --graph refseqs.dot"; + if (! -s "refseqs.dot") { + &process_cmd($cmd); + } + + # generate sequence graphs just refseqs + $cmd = "$ENV{TRINITY_HOME}/util/misc/Monarch --misc_seqs $ref_trans_fa,trinity_out_dir/inchworm.K25.L25.fa --graph refseqs_w_iworm.dot"; + if (! -s "refseqs_w_iworm.dot") { + &process_cmd($cmd); + } + + # generate sequence graphs combining all + $cmd = "$ENV{TRINITY_HOME}/util/misc/Monarch --misc_seqs $ref_trans_fa,trinity_out_dir/inchworm.K25.L25.fa,trinity_out_dir.Trinity.fasta --graph all_compare.dot"; + if (! -s "all_compare.dot") { + &process_cmd($cmd); + } + } + + + # compare refseqs to the trinity assemblies + $cmd = "$ENV{TRINITY_HOME}/util/misc/illustrate_ref_comparison.pl $ref_trans_fa trinity_out_dir.Trinity.fasta $min_per_id | tee ref_compare.ascii_illus"; + &process_cmd($cmd); + + + ## get number of transcripts reconstructed: + $cmd = "grep '>' trinity_out_dir.Trinity.fasta | wc -l"; + my $result = `$cmd`; + $result =~ s/^\s+//g; + my ($num_transcripts, @rest) = split(/\s+/, $result); + + + + # reconstruction test + $cmd = "$ENV{TRINITY_HOME}/Analysis/FL_reconstruction_analysis/FL_trans_analysis_pipeline.pl --target $ref_trans_fa --query trinity_out_dir.Trinity.fasta --no_reuse --out_prefix FL.test --allow_non_unique_mappings --min_per_length 90 --min_per_id $min_per_id | tee FL_analysis.txt"; + + &process_cmd("echo $cmd > FL.cmd"); + my @results = `$cmd`; + print @results; + + chomp @results; + shift @results; + shift @results; + $result = shift @results; + $result =~ s/^\s+//; + + my @pts = split(/\s+/, $result); + my $num_FL = $pts[2] || 0; + + my $got_all_flag = 0; + + if ($num_FL == $num_entries) { + print STDERR "-got all FL ($num_FL reconstructed / $num_entries total reconstructed).\n"; + + $got_all_flag = 1; + #print STDERR Dumper(\@pts); + } + else { + print STDERR "** missed at least one reconstructed isoform ($num_FL reconstructed / $num_entries total reconstructed).\n"; + } + + + + # cleanup really needed after all. + unless ($NO_CLEANUP) { + system("rm -rf ./trinity_out_dir"); + } + + return ($num_FL, $num_entries, $num_transcripts, $has_all_iworm_kmers, $has_all_precious_edges, $num_LR_threaded); + + +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + + +#### +sub check_inchworm_kmer_content { + my ($ref_trans_fa, $iworm_file) = @_; + + my $cmd = "$ENV{TRINITY_HOME}/Inchworm/bin/inchworm --threadFasta $ref_trans_fa --reads $iworm_file > $iworm_file.ref_kmer_check"; + &process_cmd($cmd); + + my $has_all_kmers = "YES"; + + open (my $fh, "$iworm_file.ref_kmer_check") or die $!; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my $kmer_info = $x[1]; + if ($kmer_info && $kmer_info =~ /:/) { + my ($kmer, $count) = split(/:/, $kmer_info); + if ($count == 0) { + print "Inchworm missing kmer: $kmer\n"; + $has_all_kmers = "NO"; + } + } + } + close $fh; + + return($has_all_kmers); +} + +#### +sub check_pruning { + my ($log_file, $ref_kmers_file) = @_; + + my $pruned_ref_edges = `$FindBin::Bin/util/find_pruned_edges_shouldve_kept.pl $ref_kmers_file $log_file`; + + if ($pruned_ref_edges =~ /\w/) { + print "Pruned precious edges: $pruned_ref_edges\n"; + return("NO"); + } + else { + print "no pruning of precious edges\n"; + return("YES"); + } +} + +#### +sub count_num_LR_threaded { + my ($logfile) = @_; + + # FINAL BEST PATH for LR$|ENST00000479454.1;ASZ1_mutated is [1, 228, 373, 2906, 598, 2885, 771] with total mm: 55 + # No read mapping found for: LR$|ENST00000465832.1;ASZ1_mutated + + my $num_LR_threaded = 0; + + open(my $fh, $logfile) or die "Error, cannot open file $logfile"; + while(<$fh>) { + chomp; + if (/FINAL BEST PATH for LR\$/) { + print "$_\n"; + $num_LR_threaded++; + } + elsif (/No read mapping found for: LR\$/) { + print "$_\n"; + } + } + close $fh; + + return($num_LR_threaded); +} diff --git a/99.scripts/trinity_utils/util/misc/sim_test_framework/run_Trinity_eval.sh b/99.scripts/trinity_utils/util/misc/sim_test_framework/run_Trinity_eval.sh new file mode 100644 index 0000000..4a02472 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/sim_test_framework/run_Trinity_eval.sh @@ -0,0 +1,16 @@ +#!/bin/bash + +#source /broad/software/scripts/useuse + +#reuse Perl-5.8 +#reuse .samtools-0.1.19 +#reuse GCC-4.9 + + + +CMD="`dirname $0`/run_Trinity_eval.pl $*" + +eval $CMD + +exit $? + diff --git a/99.scripts/trinity_utils/util/misc/sim_test_framework/run_simulate_reads.wgsim.pl b/99.scripts/trinity_utils/util/misc/sim_test_framework/run_simulate_reads.wgsim.pl new file mode 100644 index 0000000..f6390c6 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/sim_test_framework/run_simulate_reads.wgsim.pl @@ -0,0 +1,148 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$ENV{TRINITY_HOME}/PerlLib/"); +use Fasta_reader; +use Cwd; +use Data::Dumper; +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use List::Util qw (shuffle); + +my $help_flag; + + +my $usage = <<__EOUSAGE__; + +################################################################################ + +$0 + +################################################################################ +# +# * Required: +# +# --ref_trans|R reference transcriptome +# +# --out_dir|O output directory name +# +# --read_length default: 76 +# +# --frag_length default: 300 +# +# --depth_of_cov default: 100 +# +# +#### +# +# following wgsim options are pass-through: +# +# Options: +# -e FLOAT base error rate [0.020] +# -s INT standard deviation [50] +# -r FLOAT rate of mutations [0.0010] +# -R FLOAT fraction of indels [0.15] +# -X FLOAT probability an indel is extended [0.30] +# -S INT seed for random generator [-1] +# -A FLOAT disgard if the fraction of ambiguous bases higher than FLOAT [0.05] +# -h haplotype mode +# -Z INT strand specific mode: 1=FR, 2=RF +# -D debug mode... highly verbose +# +# +############################################################################################ + + +__EOUSAGE__ + + ; + + +my $OUT_DIR; +my $ref_trans_fa; +my $read_length = 76; +my $frag_length = 300; +my $depth_of_cov = 100; + + +&GetOptions ( 'help' => \$help_flag, + + # required + 'ref_trans|R=s' => \$ref_trans_fa, + + # optional + 'out_dir|O=s' => \$OUT_DIR, + + 'read_length=i' => \$read_length, + 'frag_length=i' => \$frag_length, + 'depth_of_cov=i' => \$depth_of_cov, + + ); + + +if ($help_flag) { + die $usage; +} + + +unless ($ref_trans_fa && $OUT_DIR) { + die $usage; +} + + + +unless ($ENV{TRINITY_HOME}) { + $ENV{TRINITY_HOME} = "$FindBin::Bin/../../trinityrnaseq/"; +} + + +main: { + + my $BASEDIR = cwd(); + + unless ($ref_trans_fa =~ /^\//) { + $ref_trans_fa = "$BASEDIR/$ref_trans_fa"; + } + + + unless (-d $OUT_DIR) { + &process_cmd("mkdir -p $OUT_DIR"); + } + chdir $OUT_DIR or die "Error, cannot cd to $OUT_DIR"; + + + my $cmd = ""; + + + # simulate reads: + $cmd = "$ENV{TRINITY_HOME}/util/misc/simulate_illuminaPE_from_transcripts.wgsim.pl --transcripts $ref_trans_fa " + . " --read_length $read_length " + . " --frag_length $frag_length " + . " --depth_of_cov 200 " + . " @ARGV "; # wgsim opts pass-through + ; + + ## todo: add mutation rate info + + &process_cmd($cmd); + +} + + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + diff --git a/99.scripts/trinity_utils/util/misc/sim_test_framework/util/find_pruned_edges_shouldve_kept.pl b/99.scripts/trinity_utils/util/misc/sim_test_framework/util/find_pruned_edges_shouldve_kept.pl new file mode 100644 index 0000000..78f3cb4 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/sim_test_framework/util/find_pruned_edges_shouldve_kept.pl @@ -0,0 +1,42 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "\n\n\tusage: $0 file.kmers bfly.log\n\n"; + +my $file_kmers = $ARGV[0] or die $usage; +my $bfly_log = $ARGV[1] or die $usage; + +my %kmers; +{ + open (my $fh, $file_kmers) or die $!; + while (<$fh>) { + chomp; + $kmers{$_} = 1; + } + close $fh; +} + +open (my $fh, $bfly_log) or die "Error, cannot open file $bfly_log"; +while (<$fh>) { + my $line = $_; + chomp; + # EDGE_PRUNING::removeLightOutEdges() removing the edge: G:W-1(V6691_D-1) GAGGCTGTGAAGAGACTGGCAGAG -> G:W-1(V9713_D-1) GAGGCTGTGAAGAGACTGGCAGAG (weight: 1.0 <= e_edge_thr: 6.550000000000001, EDGE_THR=0.05 + + if (/^EDGE_PRUNING/) { + my @x = split(/\s+/); + my $kmer_A = $x[5]; + my $kmer_B = $x[6]; + + if ($kmers{$kmer_A} && $kmers{$kmer_B}) { + print "!!\t$line"; + } + } + +} +exit(0); + + + + diff --git a/99.scripts/trinity_utils/util/misc/sim_test_framework/write_simulate_read_commands.pl b/99.scripts/trinity_utils/util/misc/sim_test_framework/write_simulate_read_commands.pl new file mode 100644 index 0000000..9d81df6 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/sim_test_framework/write_simulate_read_commands.pl @@ -0,0 +1,28 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use File::Basename; + +my $usage = "\n\n\tusage: $0 target_trans_files.list [opts ex. --wgsim ...]\n\n"; + +my $target_trans_files_file = $ARGV[0] or die $usage; +shift @ARGV; + + +main: { + + my @ref_files = `cat $target_trans_files_file`; + chomp @ref_files; + + foreach my $file (@ref_files) { + my $outdir = dirname($file); + my $cmd = "$FindBin::Bin/run_simulate_reads.wgsim.pl -R $file -O $outdir @ARGV"; + + print "$cmd\n"; + } + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/misc/simulate_illuminaPE_from_transcripts.pl b/99.scripts/trinity_utils/util/misc/simulate_illuminaPE_from_transcripts.pl new file mode 100644 index 0000000..3909e2a --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/simulate_illuminaPE_from_transcripts.pl @@ -0,0 +1,354 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; +use Nuc_translator; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use Cwd; + + +my $volcano_spacing = 25; + +my $usage = <<__EOUSAGE__; + +################################################################## +# +# Required: +# +# --transcripts file containing target transcripts in fasta format +# +# Optional: +# +# --read_length default: 76 +# +# --spacing default: 1 (simulate read from every (spacing) position) +# +# --frag_length default: 300 +# --frag_length_step default: 100 (only if max_depth > 1) +# +# --out_prefix default: 'reads' +# +# --require_proper_pairs default(off) +# +# --include_volcano_spread default(off) +# --volcano_spacing default: $volcano_spacing +# +# --error_rate default(0), for 1%, set to 0.01 +# +# --max_depth default(1), note will be double this if --include_volcano_spread is set. +# +# :: note, generates left.fa and right.fa in FR stranded format +# +# --make_fastq generate fastq instead of fasta (note, uses 'C' for all the qual scores) +# +################################################################# + +__EOUSAGE__ + + ; + + + +my $require_proper_pairs_flag = 0; + +my $transcripts; +my $read_length = 76; +my $spacing = 1; +my $frag_length = 300; +my $frag_length_step = 100; +my $help_flag; +my $out_prefix = "reads"; +my $include_volcano_spread = 0; +my $error_rate = 0; +my $MAX_DEPTH = 1; + +my $make_fastq_flag = 0; + +&GetOptions ( 'h' => \$help_flag, + 'transcripts=s' => \$transcripts, + 'read_length=i' => \$read_length, + 'spacing=i' => \$spacing, + 'frag_length=i' => \$frag_length, + 'frag_length_step=i' => \$frag_length_step, + 'out_prefix=s' => \$out_prefix, + 'require_proper_pairs' => \$require_proper_pairs_flag, + 'include_volcano_spread' => \$include_volcano_spread, + 'error_rate=f' => \$error_rate, + 'max_depth=i' => \$MAX_DEPTH, + 'make_fastq' => \$make_fastq_flag, + 'volcano_spacing=i' => \$volcano_spacing, + ); + + +if ($help_flag) { + die $usage; +} + +unless ($transcripts) { + die $usage; +} + +if ($error_rate > 0.25) { + die "Error, error rate is set to $error_rate, which exceeds a max of 0.25 "; +} + + +main: { + + my $fasta_reader = new Fasta_reader($transcripts); + print STDERR "-parsing incoming $transcripts..."; + my %read_seqs = $fasta_reader->retrieve_all_seqs_hash(); + print STDERR "done.\n"; + + my $num_trans = scalar (keys %read_seqs); + my $counter = 0; + + my $ext = ($make_fastq_flag) ? "fq" : "fa"; + + unless ($out_prefix =~ /^\//) { + $out_prefix = cwd() . "/$out_prefix"; + } + + $out_prefix = "$out_prefix.simPE_R${read_length}_F${frag_length}_FR"; + + open (my $left_ofh, ">$out_prefix.left.$ext") or die $!; + open (my $right_ofh, ">$out_prefix.right.$ext") or die $!; + + { + # write info file + open (my $ofh, ">$out_prefix.info") or die "Error, cannot write to $out_prefix.info"; + print $ofh join("\t", $transcripts, "$out_prefix.left.$ext", "$out_prefix.right.$ext") . "\n"; + close $ofh; + } + + my $FRAG_LENGTH = $frag_length; + + + foreach my $read_acc (keys %read_seqs) { + + my $seq = uc $read_seqs{$read_acc}; + + $frag_length = $FRAG_LENGTH; ## reinit + + for my $depth (1..$MAX_DEPTH) { + + $counter++; + print STDERR "\r[" . sprintf("%.2f%% = $counter/$num_trans] ", $counter/$num_trans*100); + + ## uniform dist + for (my $i = 0; $i <= length($seq); $i+=$spacing) { + + my $left_read_seq = ""; + my $right_read_seq = ""; + my $ill_acc = $read_acc . "_Ap$i-D$depth-F$frag_length-$counter"; + + my $left_start = $i; + if ($left_start >= 0) { + $left_read_seq = substr($seq, $left_start, $read_length); + } + + my $right_start = $i + $frag_length - $read_length + 1; + if ($right_start + $read_length <= length($seq)) { + $right_read_seq = substr($seq, $right_start, $read_length); + } + + if ($require_proper_pairs_flag && ! ($left_read_seq && $right_read_seq)) { next; } + + + if ($left_read_seq) { + if ($error_rate > 0) { + $left_read_seq = &introduce_errors($left_read_seq, $error_rate); + } + + &write_seq($left_ofh, "$ill_acc/1", $left_read_seq); + } + if ($right_read_seq) { + $right_read_seq = &reverse_complement($right_read_seq); + + if ($error_rate > 0) { + $right_read_seq = &introduce_errors($right_read_seq, $error_rate); + } + + &write_seq($right_ofh, "$ill_acc/2", $right_read_seq); + } + + + + } + $frag_length += $frag_length_step; + } # end depth + + + volcano_spread: + if ($include_volcano_spread) { + + ## volcano spread + for (my $i = 0; $i < length($seq) - 2 * $read_length; $i+=$volcano_spacing) { + + for (my $j = $i + 2 * $read_length; $j < length($seq) - $read_length; $j += $volcano_spacing) { + + $i += 1; + + my $ill_acc = $read_acc . "_Bp-${i}-${j}_volcano"; + + my $left_start = $i; + my $left_read_seq = substr($seq, $left_start, $read_length); + + my $right_start = $j; + + if ($left_start + $read_length >= $right_start) { next; } ## don't overlap them. + + my $right_read_seq = substr($seq, $right_start, $read_length); + + + if ($require_proper_pairs_flag && ! ($left_read_seq && $right_read_seq)) { next; } + + if ($error_rate > 0) { + $left_read_seq = &introduce_errors($left_read_seq, $error_rate); + } + + &write_seq($left_ofh, "$ill_acc/1", $left_read_seq); + + + $right_read_seq = &reverse_complement($right_read_seq); + + if ($error_rate > 0) { + $right_read_seq = &introduce_errors($right_read_seq, $error_rate); + } + + &write_seq($right_ofh, "$ill_acc/2", $right_read_seq); + + } + + + } + } + + } + + close $left_ofh; + close $right_ofh; + + print STDERR "\nDone.\n"; + + exit(0); +} + + + + +#### +sub write_seq { + my ($ofh, $acc, $seq) = @_; + + if ($make_fastq_flag) { + print $ofh join("\n", "\@$acc", $seq, "+", 'C' x length($seq)) . "\n"; + } + else { + # fasta + print $ofh join("\n", ">$acc", $seq) . "\n"; + } + return; +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + +#### +sub capture_kmer_cov_text { + my ($kmer_cov_file) = @_; + + my %kmer_cov; + + my $acc = ""; + open (my $fh, $kmer_cov_file) or die "Error, cannt open file $kmer_cov_file"; + while (<$fh>) { + chomp; + if (/>(\S+)/) { + $acc = $1; + } + else { + $kmer_cov{$acc} .= " $_"; + } + } + close $fh; + + return(%kmer_cov); +} + + +#### +sub avg { + my (@vals) = @_; + + if (scalar(@vals) == 1) { + return($vals[0]); + } + + + my $sum = 0; + foreach my $val (@vals) { + $sum += $val; + } + + my $avg = $sum / scalar(@vals); + + + return(int($avg+0.5)); +} + +#### +sub introduce_errors { + my ($sequence, $rate) = @_; + + my @seq_chars = split(//, uc $sequence); + + my $num_errors = int(length($sequence) * $rate + 0.5); + + if ($num_errors > length($sequence)) { + confess "ERROR, error rate $rate is yielding more errors than the length of the sequence itself"; + } + + my @mut_chars = qw(G A T C); + + for (1..$num_errors) { + + my $rand_pos = int(rand(length($sequence))); + + my $seq_char = $seq_chars[$rand_pos]; + + my @possible_mut_chars = grep { $_ ne $seq_char } @mut_chars; + + my $mut_char = $possible_mut_chars[ int(rand(3)) ]; + if ($mut_char eq $seq_char) { + die "Error, mut_char == seq_char, shouldn't happen"; + } + + $seq_chars[$rand_pos] = $mut_char; + + } + + my $mutated_seq = join("", @seq_chars); + + return($mutated_seq); +} + + diff --git a/99.scripts/trinity_utils/util/misc/simulate_illuminaPE_from_transcripts.wgsim.pl b/99.scripts/trinity_utils/util/misc/simulate_illuminaPE_from_transcripts.wgsim.pl new file mode 100644 index 0000000..f8a8a2d --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/simulate_illuminaPE_from_transcripts.wgsim.pl @@ -0,0 +1,176 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; +use Nuc_translator; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use Cwd; + +my $usage = <<__EOUSAGE__; + + +################################################################## + +$0 + +################################################################## +# +# Required: +# +# --transcripts file containing target transcripts in fasta format +# +# Optional: +# +# --read_length default: 76 +# +# --frag_length default: 300 +# +# --out_prefix default: 'reads' +# +# --depth_of_cov default: 100 (100x targeted base coverage) +# +#### +# +# following wgsim options are pass-through: +# +# Options: +# -e FLOAT base error rate [0.020] +# -s INT standard deviation [50] +# -r FLOAT rate of mutations [0.0010] +# -R FLOAT fraction of indels [0.15] +# -X FLOAT probability an indel is extended [0.30] +# -S INT seed for random generator [-1] +# -A FLOAT disgard if the fraction of ambiguous bases higher than FLOAT [0.05] +# -h haplotype mode +# -Z INT strand specific mode: 1=FR, 2=RF +# -D debug mode... highly verbose +# +# +############################################################################################ + + + +__EOUSAGE__ + + ; + + + +my $require_proper_pairs_flag = 0; + +my $transcripts; +my $read_length = 76; +my $frag_length = 300; +my $help_flag; +my $out_prefix = "reads"; +my $depth_of_cov = 100; + +&GetOptions ( 'help' => \$help_flag, + 'transcripts=s' => \$transcripts, + 'read_length=i' => \$read_length, + 'frag_length=i' => \$frag_length, + 'out_prefix=s' => \$out_prefix, + 'depth_of_cov=i' => \$depth_of_cov, + + + ); + + + +if ($help_flag) { + die $usage; +} + +unless ($transcripts) { + die $usage; +} + +unless ($out_prefix =~ /^\//) { + $out_prefix = cwd() . "/$out_prefix"; +} + +main: { + + my $number_reads = &estimate_total_read_count($transcripts, $depth_of_cov, $read_length); + + my $cmd = "wgsim-trans -N $number_reads -1 $read_length -2 $read_length " + . " -d $frag_length " + . " @ARGV > $out_prefix.log"; # pass-through to wgsim + + + my $token = "wgsim_R${read_length}_F${frag_length}_D${depth_of_cov}"; + if (grep { "-Z" } @ARGV) { + $token .= "_SS"; + } + + + + my $left_prefix = "$out_prefix.$token.left"; + my $right_prefix = "$out_prefix.$token.right"; + + $cmd .= " $transcripts $left_prefix.fq $right_prefix.fq"; + + &process_cmd($cmd); + + # convert to fasta format + &process_cmd("$FindBin::Bin/../support_scripts/fastQ_to_fastA.pl -I $left_prefix.fq > $left_prefix.fa"); + + &process_cmd("$FindBin::Bin/../support_scripts/fastQ_to_fastA.pl -I $right_prefix.fq > $right_prefix.fa"); + + #unlink("$left_prefix.fq", "$right_prefix.fq"); + + open(my $ofh, ">$out_prefix.$token.info") or die "Error, cannot write to file: $out_prefix.$token.info"; + print $ofh join("\t", $transcripts, "$left_prefix.fa", "$right_prefix.fa"); + close $ofh; + + + exit(0); + +} + + +#### +sub estimate_total_read_count { + my ($transcripts_fasta_file, $depth_of_cov, $read_length) = @_; + + my $sum_seq_length = 0; + my $fasta_reader = new Fasta_reader($transcripts); + while (my $seq_obj = $fasta_reader->next()) { + my $sequence = $seq_obj->get_sequence(); + $sum_seq_length += length($sequence); + } + + # DOC = num_reads * read_length * 2 / seq_length + + # so, + + # num_reads = DOC * seq_length / 2 / read_length + + my $num_reads = $depth_of_cov * $sum_seq_length / 2 / $read_length; + + $num_reads = int($num_reads + 0.5); + + return($num_reads); +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + diff --git a/99.scripts/trinity_utils/util/misc/simulate_reads_sam_and_fa.pl b/99.scripts/trinity_utils/util/misc/simulate_reads_sam_and_fa.pl new file mode 100644 index 0000000..1037ddb --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/simulate_reads_sam_and_fa.pl @@ -0,0 +1,776 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib "$ENV{TRINITY_HOME}/PerlLib"; +use Simulate::Uniform_Read_Generator; +use Overlap_info; +use GFF3_utils; +use GTF_utils; +use Gene_obj; +use Fasta_reader; +use SAM_entry; +use Data::Dumper; +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling); +use CIGAR; +use Nuc_translator; +use List::Util qw (shuffle); +use Storable qw (dclone); + +$ENV{LC_ALL} = 'C'; + + +my $usage = <<__EOUSAGE__; + +##################################################################################### +# +# Required: +# +# --gff3 reference annotations in gff3 format +# or +# --gtf reference annotations in gtf format +# +# --genome reference genome in fasta file format +# +# --frag_length mean fragment length +# +# --read_length read length from end of fragment +# +# +# Optional: +# +# --SS_lib_type if paired: RF, FR; if single: R or F +# +# --paired default (off, meaning single) +# +# --out_prefix default: 'simul' +# +#################################################################################### + + + +__EOUSAGE__ + + ; + +# --frag_length_stdev standard deviation for normal distribution of fragment lengths + + +my $gff3_file; +my $gtf_file; +my $read_length; +my $mean_frag_length; +my $frag_length_stdev = 10; # not used yet + +my $reads_per_transcript_file; +my $genome_file; +my $SS_lib_type; +my $paired_flag = 0; +my $out_prefix = "simul"; + + +&GetOptions( "gff3=s" => \$gff3_file, + "gtf=s" => \$gtf_file, + "read_length=i" => \$read_length, + "frag_length=i" => \$mean_frag_length, + "frag_length_stdev=i" => \$frag_length_stdev, + + "genome=s" => \$genome_file, + "SS_lib_type=s" => \$SS_lib_type, + "paired" => \$paired_flag, + "out_prefix=s" => \$out_prefix, + + ); + +unless ( ($gff3_file || $gtf_file) && $genome_file + && $mean_frag_length && $read_length && $frag_length_stdev) { + die $usage; +} + + +if ($SS_lib_type) { + unless ($SS_lib_type =~ /^(RF|FR|F|R)$/) { + die "Error, do not recognize SS_lib_type: [$SS_lib_type] "; + } + if ($paired_flag) { + unless ($SS_lib_type eq "RF" || $SS_lib_type eq "FR") { + die "Error, SS_lib_type: $SS_lib_type is not acceptable for paired reads."; + } + } + +} + + +if ($read_length > $mean_frag_length) { + die "Error, read length: $read_length exceeds the fragment lenth: $mean_frag_length "; +} + +main: { + + srand(); + + my $genome_sam_outfile = "$out_prefix.genome.sam"; + open (my $genome_sam_FH, ">$genome_sam_outfile") or die "Error, cannot write to $genome_sam_outfile"; + + my $trans_sam_outfile = "$out_prefix.transcriptome.sam"; + open (my $trans_sam_FH, ">$trans_sam_outfile") or die "Error, cannot write to $trans_sam_outfile"; + + + my $cdna_file = "$out_prefix.transcriptome.cdnas"; + open (my $cdna_ofh, ">$cdna_file") or die $!; + + + my $fasta_reader = new Fasta_reader($genome_file); + my %transcript_seqs = $fasta_reader->retrieve_all_seqs_hash(); + + my $reads_file = "$out_prefix.reads.fa"; + open (my $reads_ofh, ">$reads_file") or die $!; + + + my $read_counter = 0; + + ## get transcript structures: if genome, can be complex. If transcriptome, should be simple (single coordinates per transcript). + my $gene_obj_indexer_href = {}; + my $contig_to_gene_list_href; + + ## associate gene identifiers with contig id's. + if ($gff3_file) { + $contig_to_gene_list_href = &GFF3_utils::index_GFF3_gene_objs($gff3_file, $gene_obj_indexer_href); + } + else { + $contig_to_gene_list_href = >F_utils::index_GTF_gene_objs($gtf_file, $gene_obj_indexer_href); + } + + + foreach my $contig (keys %$contig_to_gene_list_href) { + + my $contig_seq = $transcript_seqs{$contig} or die "Error, no contig sequence for $contig"; + + my @gene_ids = @{$contig_to_gene_list_href->{$contig}}; + + foreach my $gene_id (@gene_ids) { + + print STDERR "\n-processing $contig :: $gene_id\n"; + + my $gene_obj = $gene_obj_indexer_href->{$gene_id}; + + + + my $orientation = $gene_obj->get_orientation(); + + + my %iso_coordsets = &get_isoform_to_coordsets($gene_obj); + my %iso_cdna_seqs = &get_isoform_seqs(\$contig_seq, \%iso_coordsets, $orientation); + + + + my @all_oriented_reads; + + foreach my $isoform_acc (keys %iso_coordsets) { + + my $coordset = $iso_coordsets{$isoform_acc} or die "Error, no coordset set for $isoform_acc"; + + my $isoform_seq = $iso_cdna_seqs{$isoform_acc} or die "Error, no cdnaseq for $isoform_acc"; + print $cdna_ofh ">$isoform_acc\n$isoform_seq\n"; + + my $read_simulator = new Simulate::Uniform_Read_Generator( { coordsets => $coordset, + mean_fragment_length => $mean_frag_length, + fragment_length_stdev => $frag_length_stdev, + read_length => $read_length, + } + ); + + print STDERR "-simulating reads for $isoform_acc\n"; + + my @paired_reads = $read_simulator->simulate_paired_reads_uniformly_across_seq(); + + + foreach my $paired_read (@paired_reads) { + + my ($left_read, $right_read) = @$paired_read; + + + if ($orientation eq '-') { + + ($left_read, $right_read) = ($right_read, $left_read); + } + + + my $read_name = "R" . sprintf("%09s", ++$read_counter); + + + my @oriented_reads; + if ($SS_lib_type) { + + # (+) and (-) are respective to the genome, not to the transcript. + + if ($SS_lib_type eq "RF") { + push (@oriented_reads, [$right_read, 1, '-', $read_name], [$left_read, 2, '+', $read_name]); + } + elsif ($SS_lib_type eq "FR") { + push (@oriented_reads, [$left_read, 1, '+', $read_name], [$right_read, 2, '-', $read_name]); + } + elsif ($SS_lib_type eq "F") { + push (@oriented_reads, [$left_read, 0, '+', $read_name]); + } + elsif ($SS_lib_type eq "R") { + push (@oriented_reads, [$right_read, 0, '-', $read_name]); + } + else { + die "Error, SS_lib_type [$SS_lib_type] not recognized"; + } + } + else { + # not strand-specific + my ($pos_1, $pos_2) = shuffle(1,2); + + if ($paired_flag) { + push (@oriented_reads, [$left_read, $pos_1, '+', $read_name], [$right_read, $pos_2, '-', $read_name]); + + } + else { + # unpaired, choose one read or the other randomly. + my @read_options = shuffle ( [$left_read, 0, '+', $read_name], [$right_read, 0, '-', $read_name]); + push (@oriented_reads, $read_options[0]); + } + } + + my %opposite = ( '+' => '-', + '-' => '+', + ); + + ## write reads to file + foreach my $oriented_read (@oriented_reads) { + my ($read, $frag_pos, $genome_orient, $name) = @$oriented_read; + my ($read_orient) = ($orientation eq '-') ? $opposite{$genome_orient} : $genome_orient; + + my $read_seq = &construct_seq_from_coordset($read, \$contig_seq, $read_orient); + my $fpos = ($frag_pos == 0) ? 1 : $frag_pos; + print $reads_ofh ">$name/$fpos\n$read_seq\n"; + } + + + push (@all_oriented_reads, [@oriented_reads]); # group pairs if paired. + + } # end of foreach paired read + } # end of foreach isoform + + + + ## examine isoform compatibility + my %compatibility_map; + + foreach my $oriented_readset (@all_oriented_reads) { + + foreach my $isoform_acc (keys %iso_coordsets) { + + my $isoform_coordset = $iso_coordsets{$isoform_acc}; + + if ( + ($paired_flag + && &Overlap_info::compatible_overlap_A_contains_B($isoform_coordset, $oriented_readset->[0]->[0]) + && &Overlap_info::compatible_overlap_A_contains_B($isoform_coordset, $oriented_readset->[1]->[0]) # both reads must be compatible with isoform + ) + || + # unpaired + ( (! $paired_flag) && &Overlap_info::compatible_overlap_A_contains_B($isoform_coordset, $oriented_readset->[0]->[0]) ) ) { + + $compatibility_map{$oriented_readset}->{$isoform_acc} = 1; + } + } + } + # end of isoform compatibility map + + + # write reads to sam file + my $read_counter = 0; + foreach my $oriented_readset (@all_oriented_reads) { + + $read_counter++; + print STDERR "\r[$read_counter] reads written. "; + + + &write_readset_to_SAM_file($oriented_readset, $genome_sam_FH, $orientation, $contig, \$contig_seq); + + foreach my $isoform (keys %{$compatibility_map{$oriented_readset}}) { + + my $iso_coordset = $iso_coordsets{$isoform}; + + my @isoform_oriented_readset; + + foreach my $read_info_aref (@$oriented_readset) { + + my @read_info = @$read_info_aref; + my $read_coordset = $read_info[0]; + + my $isoform_adjusted_read_coordset = &transform_to_isoform_coordinates($read_coordset, $iso_coordset, $orientation); + $read_info[0] = $isoform_adjusted_read_coordset; + + push (@isoform_oriented_readset, [@read_info]); + } + + my $isoform_seq = $iso_cdna_seqs{$isoform} or die "Error, no isoform sequence for $isoform";; + + &write_readset_to_SAM_file(\@isoform_oriented_readset, $trans_sam_FH, '+', $isoform, \$isoform_seq); + + + + } # end of writing isoform-level reads + + + + } # end of sam writing for genome reads + + } # end of each gene processing + + } # end of scaffold processing + + + close $genome_sam_FH; + close $trans_sam_FH; + close $cdna_ofh; + + ################################### + ## Process transcriptome SAM files + ## write to sam and bam formats. + + ## prepare transcriptome bams + my $cdna_fai = "$cdna_file.fai"; + my $cmd = "samtools faidx $cdna_file"; + &process_cmd($cmd); + + +# $cmd = "samtools view -bt $cdna_fai $trans_sam_outfile | samtools sort -n - $trans_sam_outfile.nameSorted"; +# &process_cmd($cmd); + + $cmd = "samtools view -bt $cdna_fai $trans_sam_outfile | samtools sort - $trans_sam_outfile.coordSorted"; + &process_cmd($cmd); + unlink("$trans_sam_outfile"); + +=strand_sep_trans + + if ($SS_lib_type) { + $cmd = "$FindBin::RealBin/../support_scripts/SAM_strand_separator.pl $trans_sam_outfile.coordSorted.bam $SS_lib_type"; + &process_cmd($cmd); + + foreach my $sam_file ("$trans_sam_outfile.coordSorted.bam.+.sam", "$trans_sam_outfile.coordSorted.bam.-.sam") { + + if (-s $sam_file) { + # convert to bam + my $bam_file = $sam_file; + $bam_file =~ s/sam$/bam/; + + my $cmd = "samtools view -bt $cdna_fai $sam_file > $bam_file"; + &process_cmd($cmd); + unlink($sam_file); + + $cmd = "samtools index $bam_file"; + &process_cmd($cmd); + } + } + } + +=cut + + ############################## + ## Process genome file + ## sort genome file + + # prepare genome bams + my $genome_fai = "$genome_file.fai"; + unless (-s $genome_fai) { + my $cmd = "samtools faidx $genome_file"; + &process_cmd($cmd); + } + + $cmd = "samtools view -bt $genome_fai $genome_sam_outfile | samtools sort - $genome_sam_outfile.coordSorted"; + &process_cmd($cmd); + + unlink("$genome_sam_outfile"); + + $cmd = "samtools index $genome_sam_outfile.coordSorted.bam"; + &process_cmd($cmd); + + +=strand_sep_genome + + if ($SS_lib_type) { + $cmd = "$FindBin::RealBin/../support_scripts/SAM_strand_separator.pl $genome_sam_outfile.coordSorted.bam $SS_lib_type"; + &process_cmd($cmd); + + foreach my $sam_file ("$genome_sam_outfile.coordSorted.bam.+.sam", "$genome_sam_outfile.coordSorted.bam.-.sam") { + + if (-s $sam_file) { + + # convert to bam + my $bam_file = $sam_file; + $bam_file =~ s/sam$/bam/; + + my $cmd = "samtools view -bt $genome_fai $sam_file > $bam_file"; + &process_cmd($cmd); + + $cmd = "samtools index $bam_file"; + &process_cmd($cmd); + } + + unlink("$sam_file"); + + } + } + +=cut + + exit(0); +} + + + +#### +sub compute_transcript_length { + my ($coordset) = @_; + + my $len = 0; + foreach my $coord_pair (@$coordset) { + my ($lend, $rend) = @$coord_pair; + $len += $rend - $lend + 1; + } + + return($len); +} + + + +#### +sub get_isoform_to_coordsets { + my ($gene_obj) = @_; + + my %isoform_to_coordsets; + + foreach my $isoform ($gene_obj, $gene_obj->get_additional_isoforms()) { + + my $isoform_id = $isoform->{Model_feat_name}; + + my @exons = $isoform->get_exons(); + + my @coords; + + foreach my $exon (@exons) { + + my ($lend, $rend) = sort {$a<=>$b} $exon->get_coords(); + + push (@coords, [$lend, $rend]); + } + + @coords = sort {$a->[0]<=>$b->[0]} @coords; + + $isoform_to_coordsets{$isoform_id} = \@coords; + } + + + return(%isoform_to_coordsets); + +} + + +#### +sub construct_seq_from_coordset { + my ($read, $genome_sref, $orientation) = @_; + + my $read_seq = ""; + foreach my $coords (@$read) { + + my ($lend, $rend) = @$coords; + + if ($rend > length($$genome_sref)) { + confess "Error, $rend > " . length($$genome_sref) . ", " . Dumper($read); + } + + my $seq = substr($$genome_sref, $lend-1, $rend - $lend + 1); + $read_seq .= $seq; + } + + if ($orientation eq '-') { + $read_seq = &reverse_complement($read_seq); + } + + return($read_seq); +} + + +#### +sub get_cdna_coords { + my ($read) = @_; + + my $last_cdna_pos = 0; + + my @cdna_coords; + + foreach my $coordset (@$read) { + my ($lend, $rend) = @$coordset; + my $left_cdna = $last_cdna_pos + 1; + my $right_cdna = $last_cdna_pos + $rend - $lend + 1; + push (@cdna_coords, [$left_cdna, $right_cdna]); + + $last_cdna_pos = $right_cdna; + } + + return(@cdna_coords); +} + + +#### +sub create_SAM_entry { + my ($read_name, $read_coordset, $frag_pos, $target_name, $read_seq, $orientation) = @_; + + my @fields; + $#fields = 10; + + $fields[0] = $read_name; + $fields[1] = 0; + $fields[2] = $target_name; + $fields[3] = $read_coordset->[0]->[0]; # first coordinate + $fields[4] = 255; + $fields[9] = $read_seq; + $fields[10] = 'B' x length($read_seq); + + my @cdna_coords = &get_cdna_coords($read_coordset); + + my $cigar_string = &CIGAR::construct_cigar($read_coordset, \@cdna_coords, length($read_seq)); + $cigar_string =~ s/D/N/g; + + $fields[5] = $cigar_string; + $fields[6] = '*'; + $fields[7] = 0; + $fields[8] = 0; + + my $sam = new SAM_entry(join("\t", @fields)); + + $sam->set_query_strand($orientation); + + if ($frag_pos == 1) { + $sam->set_first_in_pair(1); + } + elsif ($frag_pos == 2) { + $sam->set_second_in_pair(2); + } + + + return ($sam); +} + + + +#### +sub write_SAM_transcriptome { + my ($read_name, $read_coordset, $target_name, $FH, $read_seq, $orientation, $cdna_seq, $isoform_coordset) = @_; + + my @fields; + $#fields = 10; + + $fields[0] = $read_name; + $fields[1] = 0; + $fields[2] = $target_name; + + if ($orientation eq '-') { + $read_seq = &reverse_complement($read_seq); + } + + my $start_pos = index($cdna_seq, $read_seq); + if ($start_pos < 0) { + print STDERR "Read: " . Dumper($read_coordset) . "Transcript: " . Dumper($isoform_coordset); + confess "Error, cannot map read sequence to cdna:\nRead: $read_seq\nCDNA: $cdna_seq\n"; + } + + $fields[3] = $start_pos + 1; + + $fields[4] = 255; + $fields[9] = $read_seq; + $fields[10] = 'B' x length($read_seq); + + my @cdna_coords = &get_cdna_coords($read_coordset); + + my $cigar_string = length($read_seq) . "M"; + $fields[5] = $cigar_string; + $fields[6] = '*'; + $fields[7] = 0; + $fields[8] = 0; + + my $sam = new SAM_entry(join("\t", @fields)); + my $sam_line = $sam->toString(); + $sam_line .= "\tXS:A:+"; + + print $FH $sam_line . "\n"; + + + + return; +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + confess "Error, CMD: $cmd died with ret $ret"; + } + + return; +} + + +#### +sub parse_frag_counts { + my ($file) = @_; + + my %counts; + + open (my $fh, $file) or die "Error, cannot open file $file"; + while(<$fh>) { + chomp; + my ($acc, $count) = split(/\t/); + unless ($count =~ /^\d+$/) { + die "Error, cannot parse count from $file, entry: $_"; + } + $counts{$acc} = $count; + } + + return(%counts); +} + + + + +#### +sub write_readset_to_SAM_file { + my ($oriented_readset, $fh, $transcribed_orientation, $contig, $contig_seq_sref) = @_; + + my @sam_entries; + + foreach my $oriented_read (@$oriented_readset) { + + my ($read, $frag_pos, $read_orient, $read_name) = @$oriented_read; + my $read_seq = &construct_seq_from_coordset($read, $contig_seq_sref, '+'); # reads always in plus orientation for genome. + + if ($transcribed_orientation eq '-') { + ## flip the read orientation + $read_orient = ($read_orient eq '+') ? '-' : '+'; + } + + my $sam_entry = &create_SAM_entry($read_name, $read, $frag_pos, $contig, $read_seq, $read_orient); + push (@sam_entries, $sam_entry); + + } + + if ($paired_flag) { + $sam_entries[0]->set_paired(1); + $sam_entries[1]->set_paired(1); + + $sam_entries[0]->set_proper_pair(1); + $sam_entries[0]->set_mate_strand( $sam_entries[1]->get_query_strand()); + $sam_entries[0]->set_mate_scaffold_name( $sam_entries[1]->get_scaffold_name()); + $sam_entries[0]->set_mate_scaffold_position( $sam_entries[1]->get_scaffold_position()); + + $sam_entries[1]->set_proper_pair(1); + $sam_entries[1]->set_mate_strand( $sam_entries[0]->get_query_strand()); + $sam_entries[1]->set_mate_scaffold_name( $sam_entries[0]->get_scaffold_name()); + $sam_entries[1]->set_mate_scaffold_position( $sam_entries[0]->get_scaffold_position()); + } + + foreach my $sam_entry (@sam_entries) { + print $fh $sam_entry->toString(); + if ($SS_lib_type) { + my $XS_flag = "XS:A:$transcribed_orientation"; + print $fh "\t$XS_flag"; + } + + print $fh "\n"; + } + + return; +} + + +#### +sub transform_to_isoform_coordinates { + my ($read_coordset, $iso_coordset, $orientation) = @_; + + my @iso_coord_structs; + + @$iso_coordset = sort {$a->[0]<=>$b->[0]} @$iso_coordset; # sort by left segment boundary coordinate + + + my $cdna_len = 0; + + foreach my $iso_segment (@$iso_coordset) { + my ($lend, $rend) = @$iso_segment; + + my $cdna_lend = $cdna_len + 1; + my $cdna_rend = $cdna_len + $rend - $lend + 1; + + $cdna_len += $rend - $lend + 1; + + push (@iso_coord_structs, { genome_lend => $lend, + genome_rend => $rend, + + cdna_lend => $cdna_lend, + cdna_rend => $cdna_rend, + }); + + } + + my ($genome_left_bound, $genome_right_bound) = ($read_coordset->[0]->[0], $read_coordset->[$#$read_coordset]->[1]); + + my $cdna_left_bound = &_convert_genome_to_cdna_coord($genome_left_bound, \@iso_coord_structs); + my $cdna_right_bound = &_convert_genome_to_cdna_coord($genome_right_bound, \@iso_coord_structs); + + if ($orientation eq '-') { + # revcomp the coordinates + + ($cdna_left_bound, $cdna_right_bound) = ($cdna_len - $cdna_right_bound + 1, $cdna_len - $cdna_left_bound + 1); + } + + return ([ [$cdna_left_bound, $cdna_right_bound] ]); # single segment +} + +#### +sub _convert_genome_to_cdna_coord { + my ($genome_coord, $iso_coord_structs) = @_; + + + foreach my $struct (@$iso_coord_structs) { + + if ($genome_coord >= $struct->{genome_lend} && $genome_coord <= $struct->{genome_rend}) { + + my $cdna_coord = $struct->{cdna_lend} + $genome_coord - $struct->{genome_lend}; + + return($cdna_coord); + } + } + + die "Error, couldn't map genome coordinate ($genome_coord) to iso coordinate within: " . Dumper($iso_coord_structs); + +} + + +#### +sub get_isoform_seqs { + my ($genome_seq_sref, $iso_coordsets_href, $orientation) = @_; + + my %cdna_seqs; + + foreach my $isoform_acc (keys %$iso_coordsets_href) { + + my $iso_coordset = $iso_coordsets_href->{$isoform_acc}; + + my $cdna_seq = &construct_seq_from_coordset($iso_coordset, $genome_seq_sref, $orientation); + + $cdna_seqs{$isoform_acc} = $cdna_seq; + } + + return(%cdna_seqs); +} diff --git a/99.scripts/trinity_utils/util/misc/sixFrameTranslation.pl b/99.scripts/trinity_utils/util/misc/sixFrameTranslation.pl new file mode 100644 index 0000000..598c73d --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/sixFrameTranslation.pl @@ -0,0 +1,38 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; +use Nuc_translator; + + +my $usage = "\nn\\tusage: $0 nucFasta \n\n"; + +my $fasta_file = $ARGV[0] or die $usage; + +main: { + my $fasta_reader = new Fasta_reader($fasta_file); + while (my $seq_obj = $fasta_reader->next()) { + + my $accession = $seq_obj->get_accession(); + my $sequence = $seq_obj->get_sequence(); + my $header = $seq_obj->get_header(); + + for my $frame (1..6) { + my $translation = &translate_sequence($sequence, $frame); + + $translation =~ s/(\S{60})/$1\n/g; + chomp $translation; + + print ">$accession.F${frame}_trans $header\n$translation\n"; + } + } + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/misc/sort_fastq.pl b/99.scripts/trinity_utils/util/misc/sort_fastq.pl new file mode 100644 index 0000000..3fa7bfc --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/sort_fastq.pl @@ -0,0 +1,64 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fastq_reader; + +my $usage = "usage: $0 file.fastq\n\n"; + +my $fastq_file = $ARGV[0] or die $usage; + +main: { + + my $fastq_reader = new Fastq_reader($fastq_file); + + my $tab_tmp_file = "$fastq_file.tab"; + + open (my $ofh, ">$tab_tmp_file") or die $!; + + while (my $fq_entry = $fastq_reader->next()) { + + my $fq_record = $fq_entry->get_fastq_record(); + chomp $fq_record; + + my @lines = split(/\n/, $fq_record); + + print $ofh join("\t", @lines) . "\n"; + + } + close $ofh; + + ## sort by read name + + my $cmd = "sort -S 4G -T . -k1,1 $tab_tmp_file > $tab_tmp_file.sort"; + &process_cmd($cmd); + + unlink($tab_tmp_file); + + ## convert back to fastq file format + + $cmd = "cat $tab_tmp_file.sort | sed s/\\\\t/\\\\n/g > $fastq_file.sorted.fq"; + &process_cmd($cmd); + + unlink("$tab_tmp_file.sort"); + + exit(0); +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + diff --git a/99.scripts/trinity_utils/util/misc/splice_path_analysis/assess_intron_path_sensitivity.pl b/99.scripts/trinity_utils/util/misc/splice_path_analysis/assess_intron_path_sensitivity.pl new file mode 100644 index 0000000..ca698f0 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/splice_path_analysis/assess_intron_path_sensitivity.pl @@ -0,0 +1,268 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::RealBin/../../../PerlLib"); +use Gene_obj; +use GFF3_utils; +use BED_utils; +use Carp; +use Nuc_translator; +use Fasta_reader; + +my $usage = "\n\nusage: $0 reference.gff3,bed target.gff3,bed [genome.fa restricts to introns w/ consensus splices]\n\n"; + +my $reference_gff3_bed = $ARGV[0] or die $usage; +my $target_gff3_bed = $ARGV[1] or die $usage; +my $genome_fa = $ARGV[2]; + +my $DEBUG = 0; + +unless ($reference_gff3_bed =~ /(\.bed|\.gff3)$/) { + die $usage; +} + +unless ($target_gff3_bed =~ /(\.bed|\.gff3)$/) { + die $usage; +} + + +my %genome; +if ($genome_fa) { + + my $fasta_reader = new Fasta_reader($genome_fa); + %genome = $fasta_reader->retrieve_all_seqs_hash(); +} + + +my %reference_introns; +my %reference_combos; +my %reference_intron_to_combos; + +{ # parse reference: + + my $gene_obj_indexer_href = {}; + + my $contig_to_gene_list_href = ($reference_gff3_bed =~ /\.gff3$/) + ? &GFF3_utils::index_GFF3_gene_objs($reference_gff3_bed, $gene_obj_indexer_href) + : &BED_utils::index_BED_as_gene_objs($reference_gff3_bed, $gene_obj_indexer_href); + + foreach my $asmbl_id (sort keys %$contig_to_gene_list_href) { + + my @gene_ids = @{$contig_to_gene_list_href->{$asmbl_id}}; + + foreach my $gene_id (@gene_ids) { + my $gene_obj_ref = $gene_obj_indexer_href->{$gene_id}; + + foreach my $isoform ($gene_obj_ref, $gene_obj_ref->get_additional_isoforms()) { + + my $orientation = $isoform->get_orientation(); + + my $gene_id = $isoform->{TU_feat_name}; + my $isoform_id = $isoform->{Model_feat_name}; + + my @introns = $isoform->get_intron_coordinates(); + @introns = sort {$a->[0]<=>$b->[0]} @introns; + + if (@introns) { + my $intron_text = "$asmbl_id:"; + + my @intron_labels; + foreach my $intron (@introns) { + my ($end5, $end3) = sort {$a<=>$b} @$intron; + + $intron_text .= "_$end5-${end3}_"; + + my $intron_label = "$asmbl_id:$end5-${end3}"; + + $reference_introns{$intron_label} .= "$gene_id,$isoform_id;"; + push (@intron_labels, $intron_label); + } + + $reference_combos{$intron_text} .= "$gene_id,$isoform_id;"; + foreach my $intron_label (@intron_labels) { + $reference_intron_to_combos{$intron_label}->{$intron_text} = 1; + } + + } + } + } + } +} + + +if ($DEBUG) { + open (my $ofh, ">ref_info.txt") or die $!; + foreach my $key (keys %reference_combos) { + my $val = $reference_combos{$key}; + print $ofh "$key\t$val\n"; + } + + foreach my $key (keys %reference_introns) { + my $val = $reference_introns{$key}; + print $ofh "$key\t$val\n"; + } +} + + + + +## examine target: + +my $gene_obj_indexer_href = {}; + +my $contig_to_gene_list_href = ($target_gff3_bed =~ /\.gff3$/) + ? &GFF3_utils::index_GFF3_gene_objs($target_gff3_bed, $gene_obj_indexer_href) + : &BED_utils::index_BED_as_gene_objs($target_gff3_bed, $gene_obj_indexer_href); + +foreach my $asmbl_id (sort keys %$contig_to_gene_list_href) { + + my @gene_ids = @{$contig_to_gene_list_href->{$asmbl_id}}; + + + my $genome_seq; + if (%genome) { + $genome_seq = $genome{$asmbl_id}; + unless ($genome_seq) { + die "Error, cannot find genome sequence for [$asmbl_id]"; + } + } + + my $genome_seq_sref = \$genome_seq; + + foreach my $gene_id (@gene_ids) { + my $gene_obj_ref = $gene_obj_indexer_href->{$gene_id}; + + my $orient = $gene_obj_ref->get_orientation(); + + foreach my $isoform ($gene_obj_ref, $gene_obj_ref->get_additional_isoforms()) { + + my $orientation = $isoform->get_orientation(); + + my $gene_id = $isoform->{TU_feat_name}; + my $isoform_id = $isoform->{Model_feat_name}; + + my @introns = $isoform->get_intron_coordinates(); + + @introns = sort {$a->[0]<=>$b->[0]} @introns; + + if (@introns) { + + my $INTRONS_VALID_FLAG = 1; + + my @intron_labels; + + my $intron_text = "$asmbl_id:"; + foreach my $intron (@introns) { + my ($end5, $end3) = sort {$a<=>$b} @$intron; + + #my $intron_seq = substr($$genome_seq_sref, $end5-1, $end3-$end5+1); + #print "intron ($end5-$end3,$orient): $intron_seq\n"; + + $intron_text .= "_$end5-${end3}_"; + + my $intron_label = "$asmbl_id:$end5-${end3}"; + + if ($genome_seq) { + unless ($reference_introns{$intron_label} || + &consensus_intron_boundaries($genome_seq_sref, $orient, $end5, $end3)) { + $INTRONS_VALID_FLAG = 0; + next; + } + } + + push (@intron_labels, $intron_label); + + if (my $gene_info = $reference_introns{$intron_label}) { + print "REF\t$asmbl_id:$end5-$end3\t$gene_info\n"; + } + else { + print "NOVEL\t$asmbl_id:$end5-$end3\t$gene_id,$isoform_id\n"; + } + } + + if ($INTRONS_VALID_FLAG && scalar @introns > 1) { + + if (my $gene_info = $reference_combos{$intron_text}) { + print "REFCOMBO\t$intron_text\t$gene_info\n"; + } + else { + ## see if it's a substring of a reference path containing that intron. + my $found_as_substring = 0; + intron_subsearch: + foreach my $intron_label (@intron_labels) { + foreach my $intron_combo (keys %{$reference_intron_to_combos{$intron_label}}) { + if ($intron_combo =~ /$intron_text/) { + $found_as_substring = 1; + last intron_subsearch; + } + } + } + if ($found_as_substring) { + print "SUBrefCombo\t$intron_text\t$gene_id,$isoform_id\n"; + } + else { + print "NOVELCOMBO\t$intron_text\t$gene_id,$isoform_id\n"; + } + + } + + } + } + } + } +} + + + +exit(0); + +#### +sub consensus_intron_boundaries { + my ($genome_seq_sref, $orient, $lend, $rend) = @_; + + my $seq_len = length($$genome_seq_sref); + unless ($lend < $seq_len && $rend < $seq_len) { + print STDERR "-error, coordinates ($lend, $rend) are not within range of sequence: 1-$seq_len\n"; + return (0); + } + + my $left_splice_boundary = &get_left_splice($genome_seq_sref, $lend); + my $right_splice_boundary = &get_right_splice($genome_seq_sref, $rend); + + #print "SPLICE: $left_splice_boundary, $right_splice_boundary\n"; + + if ($orient eq '-') { + ($left_splice_boundary, $right_splice_boundary) = ($right_splice_boundary, $left_splice_boundary); + + $left_splice_boundary = &reverse_complement($left_splice_boundary); + $right_splice_boundary = &reverse_complement($right_splice_boundary); + + } + + if ($left_splice_boundary =~ /^(GT|GC)$/ && $right_splice_boundary eq "AG") { + return(1); + } + else { + return(0); + } +} + +#### +sub get_left_splice { + my ($genome_seq_sref, $coord) = @_; + + my $subseq = substr($$genome_seq_sref, $coord-1, 2); + + return(uc $subseq); +} + +#### +sub get_right_splice { + my ($genome_seq_sref, $coord) = @_; + + my $subseq = substr($$genome_seq_sref, $coord-2, 2); + + return(uc $subseq); +} diff --git a/99.scripts/trinity_utils/util/misc/splice_path_analysis/assess_intron_path_sensitivity.summarizer.pl b/99.scripts/trinity_utils/util/misc/splice_path_analysis/assess_intron_path_sensitivity.summarizer.pl new file mode 100644 index 0000000..19296cf --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/splice_path_analysis/assess_intron_path_sensitivity.summarizer.pl @@ -0,0 +1,56 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 file.intron_analysis ...\n\n"; + +my @files = @ARGV; +unless (@files) { + die "$usage\n\n"; +} + + +my @types = qw( + NOVEL + REF + NOVELCOMBO + REFCOMBO + SUBrefCombo); + +print "#analysis\t" . join("\t", @types) . "\n"; + +foreach my $intron_analysis_output (@files) { + + my %T = map { + $_ => 1 } @types; + + my %type_to_feature; + + open (my $fh, $intron_analysis_output) or die "Error, cannot open file $intron_analysis_output"; + while (<$fh>) { + chomp; + unless (/\w/) { next; } + my ($type, $intron_set, @rest) = split(/\t/); + + unless ($T{$type}) { + print STDERR "Error, do not understand type: $type\n$_\n"; + next; + } + + $type_to_feature{$type}->{$intron_set} = 1; + } + close $fh; + + + my @vals; + foreach my $type (@types) { + my $count = scalar (keys %{$type_to_feature{$type}}); + push (@vals, $count); + } + print $intron_analysis_output . "\t" . join("\t", @vals) . "\n"; + +} + +exit(0); + + diff --git a/99.scripts/trinity_utils/util/misc/splice_path_analysis/diff_splice_paths.pl b/99.scripts/trinity_utils/util/misc/splice_path_analysis/diff_splice_paths.pl new file mode 100644 index 0000000..1074d51 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/splice_path_analysis/diff_splice_paths.pl @@ -0,0 +1,160 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 splicePaths_A splicePaths_B\n\n"; + +my $fileA = $ARGV[0] or die $usage; +my $fileB = $ARGV[1] or die $usage; + + +main: { + + my %combosA = &get_combos($fileA); + my %combosB = &get_combos($fileB); + + my %introns_to_combos; + &index_combos_by_intron(\%combosA, \%introns_to_combos); + &index_combos_by_intron(\%combosB, \%introns_to_combos); + + my %seen_intron; + + foreach my $intron (keys %introns_to_combos) { + + if ($seen_intron{$intron}) { next; } + + my %paths; + &get_all_connected_paths($intron, \%introns_to_combos, \%seen_intron, \%paths); + + + + my @paths_A_not_B; + my @paths_B_not_A; + my @paths_both_A_and_B; + foreach my $path (keys %paths) { + if ($combosA{$path} && $combosB{$path}) { + push (@paths_both_A_and_B, $path); + } + elsif ($combosA{$path}) { + push (@paths_A_not_B, $path); + } + elsif ($combosB{$path}) { + push (@paths_B_not_A, $path); + } + } + + + ## filter out those that are subpaths of others. + @paths_A_not_B = &remove_subpaths(\@paths_A_not_B, [@paths_B_not_A]); + @paths_B_not_A = &remove_subpaths(\@paths_B_not_A, [@paths_A_not_B]); + + + print join("\t", $intron, + scalar(@paths_both_A_and_B), + scalar(@paths_A_not_B), + scalar(@paths_B_not_A)) . "\n"; + } + + + + exit(0); + +} + +#### +sub remove_subpaths { + my ($paths_query_aref, $paths_to_examine_as_containing_subpaths_aref) = @_; + + my @ok; + + foreach my $path (@$paths_query_aref) { + + my ($chr, $rest_path) = split(/:/, $path); + + my $found_as_subpath = 0; + + foreach my $other_path (@$paths_to_examine_as_containing_subpaths_aref) { + + if ($other_path =~ /$rest_path/) { + $found_as_subpath = 1; + last; + } + } + if (! $found_as_subpath) { + push (@ok, $path); + } + } + + return(@ok); +} + + +#### +sub get_combos { + my ($file) = @_; + + my %combos; + + open (my $fh, $file) or die $!; + while (<$fh>) { + chomp; + + my @x = split(/\t/); + + if ($x[0] =~ /COMBO/) { + $combos{$x[1]} = 1; + } + + } + + close $fh; + + return(%combos); +} + + +#### +sub index_combos_by_intron { + my ($combos_href, $introns_to_combos_href) = @_; + + foreach my $combo (keys %$combos_href) { + my @introns = &get_introns_from_path($combo); + foreach my $intron (@introns) { + push (@{$introns_to_combos_href->{$intron}}, $combo); + } + } + + return; +} + + +sub get_introns_from_path { + my ($path) = @_; + my ($chr, @introns) = split(/_+/, $path); + + foreach my $intron (@introns) { + $intron = "$chr$intron"; + } + return(@introns); +} + + +sub get_all_connected_paths { + my ($intron, $introns_to_combos_href, $seen_href, $paths_href) = @_; + + my @paths = @{$introns_to_combos_href->{$intron}}; + $seen_href->{$intron} = 1; + foreach my $path (@paths) { + if (! exists $paths_href->{$path}) { + $paths_href->{$path} = 1; + my @introns = &get_introns_from_path($path); + foreach my $intron (@introns) { + if ($seen_href->{$intron}) { next; } + &get_all_connected_paths($intron, $introns_to_combos_href, $seen_href, $paths_href); + } + } + } + + return; +} diff --git a/99.scripts/trinity_utils/util/misc/splice_path_analysis/intron_barcharter.pl b/99.scripts/trinity_utils/util/misc/splice_path_analysis/intron_barcharter.pl new file mode 100644 index 0000000..a912bbb --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/splice_path_analysis/intron_barcharter.pl @@ -0,0 +1,78 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 venn.txt\n\n"; + +my $venn = $ARGV[0] or die $usage; + +my %class_to_att_count; + +my %atts; + +open (my $fh, $venn) or die "Error, cannot open file $venn"; +while (<$fh>) { + chomp; + my ($feature, $class_list) = split(/\t/); + + my @classes = split(/,/, $class_list); + + my $att; + + my ($reference) = grep { /reference/ } @classes; + + my $count = scalar (@classes); + if ($reference) { + $count--; + $att = "reference"; + @classes = grep { $_ !~ /reference/ } @classes; + } + + + elsif ( $count == 1) { + $att = "unique"; + } + elsif ($count > 1) { + $att = "shared"; + } + else { + die "Error, have only reference...?"; + } + + $atts{$att}++; + + foreach my $class (@classes) { + $class_to_att_count{$class}->{$att}++; + } +} + +close $fh; + +my @sorted_atts = reverse sort keys %atts; + +print "#class\t" . join("\t", @sorted_atts) . "\n"; + +my @classes = sort keys %class_to_att_count; + + +foreach my $class (@classes) { + + print $class; + + foreach my $att (@sorted_atts) { + + my $count = $class_to_att_count{$class}->{$att} || 0; + print "\t$count"; + } + + print "\n"; +} + +exit(0); + + + + + + diff --git a/99.scripts/trinity_utils/util/misc/strip_fasta_header.pl b/99.scripts/trinity_utils/util/misc/strip_fasta_header.pl new file mode 100644 index 0000000..18618cb --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/strip_fasta_header.pl @@ -0,0 +1,28 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + +my $usage = "usage: $0 file.fasta\n\n"; + +my $fasta_file = $ARGV[0] or die $usage; + +main: { + + + my $fasta_reader = new Fasta_reader($fasta_file); + + while (my $seq_obj = $fasta_reader->next()) { + + my $sequence = $seq_obj->get_sequence(); + + print ">\n$sequence\n"; + } + + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/misc/tab_to_fastQ.pl b/99.scripts/trinity_utils/util/misc/tab_to_fastQ.pl new file mode 100644 index 0000000..76935ea --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/tab_to_fastQ.pl @@ -0,0 +1,20 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +while (<>) { + chomp; + my @x = split(/\t/); + unless (scalar(@x) == 3) { + die "Error, 3 fields not encountered for: $_"; + } + + print "\@$x[0]\n" + . "$x[1]\n" + . "+\n" + . "$x[2]\n"; +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/misc/tab_to_fasta.pl b/99.scripts/trinity_utils/util/misc/tab_to_fasta.pl new file mode 100644 index 0000000..7351c41 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/tab_to_fasta.pl @@ -0,0 +1,24 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +main: { + while () { + unless (/\w/) { next; } + chomp; + my @x = split (/\t/); + my $seq = pop @x; + + # make fasta + $seq =~ s/(\S{60})/$1\n/g; + chomp $seq; + + print ">" . join (" ", @x) . "\n$seq\n"; + + } + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/tblastn_wrapper.pl b/99.scripts/trinity_utils/util/misc/tblastn_wrapper.pl new file mode 100644 index 0000000..d916c51 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/tblastn_wrapper.pl @@ -0,0 +1,41 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 db query [opts]\n\n"; + +unless (@ARGV) { + die $usage; +} +my $db = $ARGV[0] or die $usage; +my $query = $ARGV[1] or die $usage; + +shift @ARGV; +shift @ARGV; + +main: { + + my $cmd = "makeblastdb -in $db -dbtype nucl"; + &process_cmd($cmd) unless (-s "$db.nin"); # only build it once. + + $cmd = "tblastn -db $db -query $query -seg no @ARGV"; + &process_cmd($cmd); + + exit(0); +} + +#### +sub process_cmd { + my ($cmd) = @_; + + my $ret = system($cmd); + if ($ret) { + die "Error, CMD: $cmd died with ret $ret"; + } + + return; +} + + + diff --git a/99.scripts/trinity_utils/util/misc/testUnlimitStacksize.pl b/99.scripts/trinity_utils/util/misc/testUnlimitStacksize.pl new file mode 100644 index 0000000..d6618c0 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/testUnlimitStacksize.pl @@ -0,0 +1,80 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +no strict 'subs'; +#use BSD::Resource; + +main: { + + print "\n\nBefore unlimit stack space:"; + system("bash -c ulimit -a"); + print "Attempting to unlimit stack space..."; + try_unlimit(); + + + + + print "After unlimit stack space: "; + system("bash -c ulimit -a"); + system("ls"); + my $ls = `ls`; + print "$ls\n"; + + exit(0); +} + + +sub try_unlimit { + #eval " + # use BSD::Resource; + # setrlimit(RLIMIT_STACK, RLIM_INFINITY, RLIM_INFINITY);"; + + + my $unset_stacksize_code = "use BSD::Resource;" + . "setrlimit(RLIMIT_STACK, RLIM_INFINITY, RLIM_INFINITY);" + . "my (\$soft, \$hard) = getrlimit(RLIMIT_STACK);" + . "print \"stack_soft: \$soft, stack_hard: \$hard \";" + . "if (\$soft != -1) { die \"Couldn't unset stacksize\";}"; + + + eval($unset_stacksize_code); + + if( $@ ) { + warn <<"EOF"; + + $@ + + Unable to set unlimited stack size. Please install the BSD::Resource + Perl module to allow this script to set the stack size, or set it + yourself in your shell before running Trinity (ignore this warning if + you have set the stack limit in your shell). See the following URL for + more information: + + http://trinityrnaseq.sourceforge.net/trinity_faq.html#ques_E + +EOF +; + } + else { + + + print "Successfully set unlimited stack size.\n"; + print "###################################\n\n"; + + + + } + + ## verify + print "\n\n"; + eval("use BSD::Resource;" + ."my (\$soft, \$hard) = getrlimit(RLIMIT_STACK);" + ."print \"Verifying: stack_soft: \$soft, stack_hard: \$hard \";" ); + + + + + return; +} + diff --git a/99.scripts/trinity_utils/util/misc/transcript_coverage_UTR_trimmer.pl b/99.scripts/trinity_utils/util/misc/transcript_coverage_UTR_trimmer.pl new file mode 100644 index 0000000..89decf2 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/transcript_coverage_UTR_trimmer.pl @@ -0,0 +1,194 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use threads; + +use FindBin; +use Getopt::Long qw(:config no_ignore_case bundling); +use lib ("$FindBin::RealBin/../../PerlLib"); +use WigParser; +use Fasta_reader; +use Statistics::Descriptive; + +my $usage = <<_EOUSAGE_; + +######################################################################################################## +# +# Required: +# +# --coord_sorted_SAM coordinate-sorted SAM file. +# --transcripts transcripts in fasta file format +# +# *If Strand-specific, specify: +# --SS_lib_type library type: if single: F or R, if paired: FR or RF +# --trim_pct_median percent of median coverage value to set as upper threshold for end-trimming (default: 10) +# +# --no_trim just provide the coverage info +# +######################################################################################################## + + +_EOUSAGE_ + + ; + + +my $help_flag; + +#my $partition_join_size; +my $SAM_file; +my $SS_lib_type = ""; +my $transcripts_fasta_file = ""; + +my $trim_pct_median = 10; +my $NO_TRIM = 0; + +&GetOptions ( 'h' => \$help_flag, + + 'coord_sorted_SAM=s' => \$SAM_file, + 'SS_lib_type=s' => \$SS_lib_type, + 'transcripts=s' => \$transcripts_fasta_file, + 'trim_pct_median=i' => \$trim_pct_median, + 'no_trim' => \$NO_TRIM, + ); + + +if ($help_flag) { + die $usage; +} + +unless ( + $SAM_file + && + $transcripts_fasta_file + ) { + die $usage; +} + +if ($SS_lib_type && $SS_lib_type !~ /^(F|R|FR|RF)$/) { + die "Error, invalid --SS_lib_type, only F, R, FR, or RF are possible values"; +} + +my $UTIL_DIR = "$FindBin::RealBin/"; + +main: { + + my $sam_file_to_process; + + if ($SS_lib_type) { + + $sam_file_to_process = "$SAM_file.+.sam"; + my $cmd = "$UTIL_DIR/SAM_strand_separator.pl $SAM_file $SS_lib_type"; + &process_cmd($cmd) unless (-s $sam_file_to_process); + + } + else { + $sam_file_to_process = $SAM_file; + } + + ## define fragments + my $cmd = "$UTIL_DIR/SAM_to_frag_coords.pl --sam $sam_file_to_process --min_insert_size 1 --max_insert_size 1000 "; ## writes file: $sam_file.frag_coords + &process_cmd($cmd) unless (-s "$sam_file_to_process.frag_coords"); + + ## define coverage + $cmd = "$UTIL_DIR/fragment_coverage_writer.pl $sam_file_to_process.frag_coords > $sam_file_to_process.frag_coverage.wig"; + &process_cmd($cmd) unless (-s "$sam_file_to_process.frag_coverage.wig"); + + my $wig_parser = new WigParser("$sam_file_to_process.frag_coverage.wig"); + + my $fasta_reader = new Fasta_reader($transcripts_fasta_file); + + while (my $seq_obj = $fasta_reader->next()) { + my $acc = $seq_obj->get_accession(); + my $sequence = $seq_obj->get_sequence(); + + + my $trim_5p_pos = "NA"; + my $trim_3p_pos = "NA"; + my $median = 0; + my @cov = (); + + eval { + # throws exception if there's no read coverage + + @cov = $wig_parser->get_wig_array($acc); + #print ">$acc\n$sequence\n" . join(" ", @cov) . "\n"; + + my $stat = Statistics::Descriptive::Full->new(); + $stat->add_data(@cov); + $median = $stat->median(); + #print "Median: $median\n"; + + my $trim_thresh = int($trim_pct_median/100 * $median + 0.5); + if ($trim_thresh == 0) { + $trim_thresh = 1; + } + #print "Trim_thresh: $trim_thresh\n"; + + + + + unless ($NO_TRIM) { + + $trim_5p_pos = &trim_5p(\@cov, $trim_thresh); + $trim_3p_pos = &trim_3p(\@cov, $trim_thresh); + } + }; + + print join("\t", $acc, "$trim_5p_pos-$trim_3p_pos", $median, $sequence, join(" ", @cov) ) . "\n"; + + } + + exit(0); + + +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + + if ($ret) { + die "Error, command $cmd died with ret $ret"; + } + + return; +} + + +#### +sub trim_5p { + my ($cov_aref, $threshold) = @_; + + for (my $i = 0; $i <= $#$cov_aref; $i++) { + my $cov = $cov_aref->[$i]; + if ($cov >= $threshold) { + return($i+1); + } + } + + return(-1); # bad sequence + +} + +#### +sub trim_3p { + my ($cov_aref, $threshold) = @_; + + for (my $i = $#$cov_aref; $i >=0; $i--) { + my $cov = $cov_aref->[$i]; + if ($cov >= $threshold) { + return($i+1); + } + } + + return(-1); # bad sequence + +} diff --git a/99.scripts/trinity_utils/util/misc/transcript_fasta_to_ORF_pics.pl b/99.scripts/trinity_utils/util/misc/transcript_fasta_to_ORF_pics.pl new file mode 100644 index 0000000..a4b95e3 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/transcript_fasta_to_ORF_pics.pl @@ -0,0 +1,171 @@ +#!/usr/bin/env perl + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use strict; +use warnings; +use Getopt::Std; +use Fasta_reader; +use Longest_orf; +use List::Util qw (min max); + +use Bio::Graphics; +use Bio::SeqFeature::Generic; + +my $usage = "\n\nusage: $0 transcripts.fasta\n\n"; + +my $transcripts_fasta_file = $ARGV[0] or die $usage; +my $min_prot_len = 50; + + +main: { + + my $fasta_reader = new Fasta_reader($transcripts_fasta_file); + + + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + my $sequence = $seq_obj->get_sequence(); + + my $seq_length = length($sequence); + + + my $panel = Bio::Graphics::Panel->new( + -length => $seq_length, + -width => 800, + -pad_left => 10, + -pad_right => 10, + ); + + my $full_length = Bio::SeqFeature::Generic->new( + -start => 1, + -end => $seq_length, + -strand => 1, + -display_name => $acc, + ); + + + + ## add ticker + $panel->add_track($full_length, + -glyph => 'arrow', + -tick => 2, + -fgcolor => 'black', + -double => 1, + + ); + + ## add feature + $panel->add_track($full_length, + -glyph => 'transcript2', + -bgcolor => 'black', + -label => 1, + ); + + + my $longest_orf_finder = new Longest_orf(); + $longest_orf_finder->allow_5prime_partials(); + $longest_orf_finder->allow_3prime_partials(); + + my @orf_structs = $longest_orf_finder->capture_all_ORFs($sequence); + + + + + @orf_structs = reverse sort {$a->{length}<=>$b->{length}} @orf_structs; + + + + + my $top_track = $panel->add_track(-glyph => 'transcript2', + -label => 1, + -strand_arrow => 1, + -bgcolor => 'blue', + + ); + + my $bottom_track = $panel->add_track(-glyph => 'transcript2', + -label => 1, + -strand_arrow => 1, + -bgcolor => 'red', + ); + + + my $pep_file = "$acc.orfs.pep"; + my $cds_file = "$acc.orfs.cds"; + + open (my $pep_ofh, ">$pep_file") or die "Error, cannot write to $pep_file"; + open (my $cds_ofh, ">$cds_file") or die "Error, cannot write to $cds_file"; + + foreach my $orf (@orf_structs) { + + my $start = $orf->{start}; + my $stop = $orf->{stop}; + + if ($stop <= 0) { $stop += 3; } # edge issue + + my $length = int($orf->{length}/3); + if ($length < $min_prot_len) { next; } + + my $orient = $orf->{orient}; + my $protein = $orf->{protein}; + my $cds = $orf->{sequence}; + + + my $frame = $start % 3; + if ($frame == 0) { + $frame = 3; + } + + my $orf_name = "$acc.$start-$stop.F$frame"; + print $pep_ofh ">$orf_name\n$protein\n"; + print $cds_ofh ">$orf_name\n$cds\n"; + + + if ($orient eq '+') { + + + my $feature = Bio::SeqFeature::Generic->new( + -start => $start, + -end => $stop, + -display_name => "orf: $start-$stop (F:$frame)", + -strand => 1, + + ); + $top_track->add_feature($feature); + + } + else { + # minus strand + + $frame = -1 * $frame; + + my $feature = Bio::SeqFeature::Generic->new( + -start => $start, + -end => $stop, + -display_name => "orf: $start-$stop (F:$frame)", + -strand => -1, + + ); + $bottom_track->add_feature($feature); + } + } + close $pep_ofh; + close $cds_ofh; + + + my $img_filename = "$acc.orf.png"; + print STDERR "-writing: $img_filename\n"; + open (my $ofh, ">$img_filename") or die "Error, cannot write to file $img_filename"; + binmode($ofh); + print $ofh $panel->png(); + close $ofh; + } + + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/misc/transcript_gff3_to_bed.pl b/99.scripts/trinity_utils/util/misc/transcript_gff3_to_bed.pl new file mode 100644 index 0000000..f03486b --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/transcript_gff3_to_bed.pl @@ -0,0 +1,103 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; + +use lib ("$FindBin::RealBin/../../PerlLib"); +use Gene_obj; + +my $usage = "usage: $0 alignments.gff3\n\n"; + +my $trans_gff3_file = $ARGV[0] or die $usage; + + +main: { + + my %genome_trans_to_coords; + + open (my $fh, $trans_gff3_file) or die "Error, cannot open file $trans_gff3_file"; + while (<$fh>) { + chomp; + + unless (/\w/) { next; } + + my @x = split(/\t/); + + unless (scalar (@x) >= 8 && $x[8] =~ /ID=/) { + print STDERR "ignoring line: $_\n"; + next; + } + + my $scaff = $x[0]; + my $type = $x[2]; + my $lend = $x[3]; + my $rend = $x[4]; + + my $orient = $x[6]; + + my $info = $x[8]; + + my @parts = split(/;/, $info); + my %atts; + foreach my $part (@parts) { + $part =~ s/^\s+|\s+$//; + $part =~ s/\"//g; + my ($att, $val) = split(/=/, $part); + + if (exists $atts{$att}) { + die "Error, already defined attribute $att in $_"; + } + + $atts{$att} = $val; + } + + my $gene_id = $atts{ID} or die "Error, no gene_id at $_"; + my $trans_id = $atts{Target} or die "Error, no trans_id at $_"; + { + my @pieces = split(/\s+/, $trans_id); + $trans_id = shift @pieces; + } + + my ($end5, $end3) = ($orient eq '+') ? ($lend, $rend) : ($rend, $lend); + + $genome_trans_to_coords{$scaff}->{$gene_id}->{$trans_id}->{$end5} = $end3; + + } + + + ## Output genes in gff3 format: + + foreach my $scaff (sort keys %genome_trans_to_coords) { + + my $genes_href = $genome_trans_to_coords{$scaff}; + + foreach my $gene_id (keys %$genes_href) { + + my $trans_href = $genes_href->{$gene_id}; + + foreach my $trans_id (keys %$trans_href) { + + my $coords_href = $trans_href->{$trans_id}; + + my $gene_obj = new Gene_obj(); + + $gene_obj->{TU_feat_name} = $gene_id; + $gene_obj->{Model_feat_name} = $trans_id; + $gene_obj->{com_name} = "$gene_id $trans_id"; + + $gene_obj->{asmbl_id} = $scaff; + + $gene_obj->populate_gene_object($coords_href, $coords_href); + + print $gene_obj->to_BED_format(); + + } + } + } + + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/misc/transdecoder_pep_to_false_fusion_finder.pl b/99.scripts/trinity_utils/util/misc/transdecoder_pep_to_false_fusion_finder.pl new file mode 100644 index 0000000..1722c70 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/transdecoder_pep_to_false_fusion_finder.pl @@ -0,0 +1,67 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +## Just identifies and counts transcripts that are predicted to encode multiple ORFs + +# of those that encode an ORF of min length, how many encode multiple ORFs of min length? + +my $usage = "\n\nusage: $0 transdecoder.pep [min_prot_length=300]\n\n"; + +my $transdecoder_pep_file = $ARGV[0] or die $usage; +my $min_prot_length = $ARGV[1] || 300; + + + +main: { + + # example transdecoder pep header + # m.83 g.83 ORF g.83 m.83 type:complete len:773 (+) CUFF.100.1:5314-7632(+) + + my %acc_orf_counter; + + open (my $fh, $transdecoder_pep_file) or die $!; + while (<$fh>) { + chomp; + if (/^>/) { + /\s(\S+):\d+-\d+\([+-]\)/ or die "Error, cannot extract transcript accession from $_"; + my $trans_acc = $1; + + /\slen:(\d+)\s/ or die "Error, cannot extract len value from $_"; + my $len = $1; + + if ($len >= $min_prot_length) { + $acc_orf_counter{$trans_acc}++; + } + } + } + close $fh; + + ## determine number of putative chimeras + my $num_trans = 0; + my $num_chims = 0; + + foreach my $acc (keys %acc_orf_counter) { + + $num_trans++; + + if ($acc_orf_counter{$acc} > 1) { + + $num_chims++; + + } + + } + + my $pct_chims = sprintf("%.2f", $num_chims/$num_trans*100); + + print "#trans\tchims\t\%chims\n"; + print join("\t", $num_trans, $num_chims, $pct_chims) . "\n"; + + + exit(0); + + + +} diff --git a/99.scripts/trinity_utils/util/misc/trinity_component_distribution.pl b/99.scripts/trinity_utils/util/misc/trinity_component_distribution.pl new file mode 100644 index 0000000..99f7cd3 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/trinity_component_distribution.pl @@ -0,0 +1,90 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use POSIX qw (ceil); +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + +my $usage = "usage: $0 Trinity.fasta [length_bin_size=100] [out_prefix='dist']\n\n"; + +my $trinity_fasta = $ARGV[0] or die $usage; +my $length_bin_size = $ARGV[1] || 100; +my $out_prefix = $ARGV[2] || "dist"; + + +main: { + + + my %comp_counter; + my %length_bin_counter; + + my $fasta_reader = new Fasta_reader($trinity_fasta); + + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + + my $sequence = $seq_obj->get_sequence(); + my $sequence_length = length($sequence); + + my $bin = ceil($sequence_length/$length_bin_size); + + $length_bin_counter{$bin}++; + + my $comp_name; + + if ($acc =~ /^(.*comp\d+_c\d+)/) { + $comp_name = $1; + } + elsif ($acc =~ /^(.*c\d+_g\d+)/) { + $comp_name = $1; + } + else { + die "Error, cannot get component/gene id from acc: $acc"; + } + + $comp_counter{$comp_name}++; + } + + ## get component dist + my %comp_size_counter; + foreach my $count (values %comp_counter) { + $comp_size_counter{$count}++; + } + + ## write component dist + open (my $ofh, ">$out_prefix.comp_sizes.txt") or die $!; + print $ofh join("\t", "transPerComp", "num_components") . "\n"; + foreach my $comp_size (sort {$a<=>$b} keys %comp_size_counter) { + + my $comp_size_count = $comp_size_counter{$comp_size}; + print $ofh join("\t", $comp_size, $comp_size_count) . "\n"; + } + close $ofh; + + ## write length distribution + open ($ofh, ">$out_prefix.trans_lengths.txt") or die $!; + print $ofh join("\t", "#length_bin", "count_trans") . "\n"; + foreach my $length_bin (sort {$a<=>$b} keys %length_bin_counter) { + + my $length = $length_bin * $length_bin_size; + + my $count = $length_bin_counter{$length_bin}; + + print $ofh join("\t", $length, $count) . "\n"; + } + close $ofh; + + + print STDERR "\n\nDone. See files $out_prefix.comp_sizes.txt and $out_prefix.trans_lengths.txt\n\n"; + + + exit(0); +} + + + + + diff --git a/99.scripts/trinity_utils/util/misc/trinity_trans_matrix_to_rep_trans_gene_matrix.pl b/99.scripts/trinity_utils/util/misc/trinity_trans_matrix_to_rep_trans_gene_matrix.pl new file mode 100644 index 0000000..b8951d8 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/trinity_trans_matrix_to_rep_trans_gene_matrix.pl @@ -0,0 +1,82 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +## sums up reads per gene, but uses the single isoform with the highest sum counts as the representative entry in the reported 'gene' matrix. + +my $usage = "usage: $0 trinity_trans_matrix > trinity_rep_trans_matrix\n\n"; + +my $trans_matrix_file = $ARGV[0] or die $usage; + +main: { + + open(my $fh, $trans_matrix_file) or die $!; + my $header = <$fh>; + + print $header; + + my %data; + while (<$fh>) { + chomp; + my $line = $_; + my @x = split(/\t/); + my $acc = $x[0]; + + #c1000023_g1_i3 + + my $gene_id; + if ($acc =~ /^([\w_]+_g\d+)_i\d+/) { + $gene_id = $1; + } + unless ($gene_id) { + die "Error, cannot extract gene identifier for $acc"; + } + + push (@{$data{$gene_id}}, $line); + } + close $fh; + + foreach my $acc (keys %data) { + my @lines = @{$data{$acc}}; + if (scalar @lines > 0) { + my $rep_line = &get_rep(@lines); + print "$rep_line\n"; + } + else { + my $rep_line = shift @lines; + print "$rep_line\n"; + } + } + + exit(0); +} + +### +sub get_rep { + my @lines = @_; + + my @vals; + my $best_rowsum = -1; + my $best_acc = ""; + + foreach my $line (@lines) { + my @x = split(/\t/, $line); + my $acc = shift @x; + my $sum = 0; + for (my $i = 0; $i <= $#x; $i++) { + my $val = $x[$i]; + $sum += $val; + $vals[$i] += $val; + } + if ($sum > $best_rowsum) { + $best_rowsum = $sum; + $best_acc = $acc; + + } + } + + my $line = join("\t", $best_acc, @vals); + return($line); +} + diff --git a/99.scripts/trinity_utils/util/misc/try_estimate_TPM_filtering_threshold.Rscript b/99.scripts/trinity_utils/util/misc/try_estimate_TPM_filtering_threshold.Rscript new file mode 100644 index 0000000..36cc221 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/try_estimate_TPM_filtering_threshold.Rscript @@ -0,0 +1,157 @@ +#!/usr/bin/env Rscript + + +## Identifies the approximate number of most relevant transcripts based on finding the point representing the elbow in the curve of expression ~ ordered transcript + + +main = function () { + + + suppressPackageStartupMessages(library("argparse")) + suppressPackageStartupMessages(library("tidyverse")) + + + + parser = ArgumentParser() + parser$add_argument("--E_inputs", help="file.isoform.TMM.EXPR.matrix.E-inputs", required=TRUE, nargs=1) + parser$add_argument("--out_pdf", help="name for output pdf filename", required=FALSE, nargs=1, default="estimate_TPM_threshold.pdf") + args = parser$parse_args() + + E_inputs_filename = args$E_inputs + output_pdf_filename = args$out_pdf + + message("-parsing:", E_inputs_filename) + + if (grepl(".gz$", E_inputs_filename)) { + data = read.table(gzfile(E_inputs_filename), header=T, com='', sep="\t", stringsAsFactors = F) + } else { + data = read.table(E_inputs_filename, header=T, com='', sep="\t", stringsAsFactors = F) + } + + orig_data = data + + data = data %>% filter(max_expr_over_samples > 0.05 & max_expr_over_samples <= 2^5) # must have some evidence of expression + + + if (nrow(data) < 1000) { + message("Too few data points in the low-expression range to perform a meaningful analysis here.") + quit(save = "no", status = 0, runLast = FALSE) + } + + data = data %>% arrange(desc(max_expr_over_samples)) %>% mutate(r=row_number()) + + data = data %>% mutate(logexpr = log2(max_expr_over_samples+1)) + + normalize = function(x) { + + x = unlist(x) + min_x = min(x) + max_x = max(x) + + + vals = (x-min_x)/(max_x-min_x) + return(vals) + } + + + data$x = normalize(list(data$logexpr)) + data$y = normalize(list(data$r)) + + + + message("-performing elbow analysis to select threshold.") + get_dist_point_line <- function(point, + line_coord1, + line_coord2) { + # derived from https://github.com/ropensci/pathviewr/blob/HEAD/R/analytical_functions.R + ## Compute + v1 <- line_coord1 - line_coord2 + v2 <- point - line_coord1 + m <- cbind(v1, v2) + dist <- abs(det(m)) / sqrt(sum(v1 * v1)) + + ## export + return(dist) + } + + + # get a line from the first and last point of the elbow curve + data = data %>% arrange(x) + first_pt = data %>% head(n=1) + last_pt = data %>% tail(n=1) + xs = c(first_pt$x, last_pt$x) + ys = c(first_pt$y, last_pt$y) + line = lm(ys ~ xs) + coeffs = coefficients(line) + intercept = coeffs[1] + slope = coeffs[2] + + dist_to_line_df = do.call(rbind, apply(data.frame(x=data$x, y=data$y), 1, function(row) { + + x = row[1] + y = row[2] + + point_dist = get_dist_point_line(c(x,y), c(first_pt$x, first_pt$y), c(last_pt$x, last_pt$y)) + + return(data.frame(dist=point_dist)) + + }) ) + + data = bind_cols(data, dist_to_line_df) + + # define threshold as that transcript with greatest distance to the line (elbow transcript) + threshold_entry = data %>% arrange(desc(dist)) %>% head(n=1) + + tpm_x = threshold_entry$max_expr_over_samples + + + + # compute some stats based on that threshold + filtered_data = orig_data %>% filter(max_expr_over_samples >= tpm_x) %>% arrange(desc(length)) + sum_length = sum(filtered_data$length) + filtered_data = filtered_data %>% mutate(cumsum_len = cumsum(length)) + half_length = sum_length / 2 + N50_entry = filtered_data %>% filter(cumsum_len <= half_length) %>% filter(row_number() == n()) + Ex = threshold_entry %>% pull(X.Ex) + N50 = N50_entry %>% pull(length) + logexpr_x = threshold_entry$logexpr + + # ensure line is above the elbow + line_y = threshold_entry$x * slope + intercept + + + pdf(output_pdf_filename) + p = data %>% ggplot(aes(x=logexpr, y=r)) + geom_point() + + if (line_y < threshold_entry$y) { + message("Not finding a lower elbow curve, cannot estimate min threshold") + + } + else { + + num_transcripts = data %>% filter(max_expr_over_samples >= tpm_x) %>% nrow() + + message("-Number of transcripts >= tpm threshold: ", num_transcripts) + message("-Fraction of expression data represented: ", Ex) + message("-Contig N50 based on these transcripts: ", N50) + message("-selected TPM threshold: ", tpm_x) + + p = p + geom_vline(xintercept=logexpr_x, color='red') + + annotate("text", x=threshold_entry$logexpr, y=num_transcripts, + label=paste0(" # transcripts =", num_transcripts, " Ex=", Ex, " N50=", N50, " TPMthresh=", tpm_x), hjust=0) + + } + + plot(p) + + + message("-Done. See pdf: estimate_TPM_threshold.pdf") + + quit(save = "no", status = 0, runLast = FALSE) + +} + + +if (length(sys.calls())==0) { + main() +} diff --git a/99.scripts/trinity_utils/util/misc/validate_fastqs.py b/99.scripts/trinity_utils/util/misc/validate_fastqs.py new file mode 100644 index 0000000..2e4e3c9 --- /dev/null +++ b/99.scripts/trinity_utils/util/misc/validate_fastqs.py @@ -0,0 +1,103 @@ +#!/usr/bin/env python3 +# encoding: utf-8 + +import os, sys, re +import logging +import argparse +import gzip +from collections import defaultdict + +def main(): + + parser = argparse.ArgumentParser(description="validate fastqs", formatter_class=argparse.ArgumentDefaultsHelpFormatter) + + + parser.add_argument("--left_fq", required=True, type=str, help="left fastq file") + + parser.add_argument("--right_fq", required=False, type=str, help="right fastq file") + + args = parser.parse_args() + + left_fq = args.left_fq + right_fq = args.right_fq + + + + + left_fq_iterator = fastq_iterator(left_fq) + right_fq_iterator = None + if right_fq: + right_fq_iterator = fastq_iterator(right_fq) + + + + counter = 0 + + for left_read_tuple in left_fq_iterator: + left_readname, left_readseq, left_L3, left_quals = left_read_tuple + left_readseq_len = len(left_readseq) + left_quals_len = len(left_quals) + + counter += 1 + if counter % 10000 == 0: + sys.stderr.write(f"\r[{counter}] ") + + + assert left_readseq_len == left_quals_len, f"Error, left seqlen and quals len dont match:\n{left_readname}\n{left_readseq}\n{left_L3}\n{left_quals}\n" + + + if right_fq_iterator is not None: + + right_read_tuple = next(right_fq_iterator) + right_readname, right_readseq, right_L3, right_quals = right_read_tuple + right_readseq_len = len(right_readseq) + right_quals_len = len(right_quals) + + + assert right_readseq_len == right_quals_len, f"Error, right seqlen and quals len dont match:\n{right_readname}\n{right_readseq}\n{right_L3}\n{right_quals}\n" + + assert core_readname(left_readname) == core_readname(right_readname), f"Error, fastq pair read names aren't consistent: {left_readname} vs. {right_readname}" + + + print("fastq(s) validate") + + sys.exit(0) + + + +def core_readname(readname): + + core_readname = readname.split(" ")[0] + + core_readname = re.sub("/[12]$", "", core_readname) + + return core_readname + + + +def fastq_iterator(fastq_filename): + + if re.search(".gz$", fastq_filename): + fh = gzip.open(fastq_filename, 'rt', encoding='utf-8') + else: + fh = open(fastq_filename, 'rt', encoding='utf-8') + + have_records = True + while have_records: + readname = next(fh).rstrip() + readseq = next(fh).rstrip() + L3 = next(fh).rstrip() + quals = next(fh).rstrip() + + yield (readname, readseq, L3, quals) + + if not readname: + break + + return + + + + +if __name__=='__main__': + main() diff --git a/99.scripts/trinity_utils/util/retrieve_sequences_from_fasta.pl b/99.scripts/trinity_utils/util/retrieve_sequences_from_fasta.pl new file mode 100644 index 0000000..9b1467d --- /dev/null +++ b/99.scripts/trinity_utils/util/retrieve_sequences_from_fasta.pl @@ -0,0 +1,53 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "\n\n\tusage: $0 acc_list_file.txt target_db.fasta\n\n"; + +my $acc_list_file = $ARGV[0] or die $usage; +my $target_db = $ARGV[1] or die $usage; + + +main: { + + my $samtools = `sh -c "command -v samtools"`; + unless ($samtools =~ /\w/) { + die "Error, need samtools in your PATH setting."; + } + chomp $samtools; + + my @accs = `cat $acc_list_file`; + chomp @accs; + + if (! -s "$target_db.fasta.fai") { + my $cmd = "$samtools faidx $target_db"; + my $ret = system $cmd; + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + } + + my $ret = 0; + + foreach my $acc (@accs) { + $acc =~ s/\s//g; + + unless ($acc =~ /\w/) { next; } + + my $cmd = "$samtools faidx $target_db \"$acc\""; + + my $result = `$cmd`; + if ($result) { + print $result; + } + else { + print STDERR "No entry retrieved for acc: $acc\n"; + $ret = 1; + } + } + + exit($ret); + +} + diff --git a/99.scripts/trinity_utils/util/sift_bam_max_cov.pl b/99.scripts/trinity_utils/util/sift_bam_max_cov.pl new file mode 100644 index 0000000..259524d --- /dev/null +++ b/99.scripts/trinity_utils/util/sift_bam_max_cov.pl @@ -0,0 +1,160 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use Getopt::Long qw(:config posix_default no_ignore_case bundling pass_through); +use FindBin; +use lib ("$FindBin::Bin/../PerlLib"); +use SAM_entry; + +my $help_flag; + +my $usage = <<__EOUSAGE__; + +########################################### +# +# Required: +# +# --bam bam file +# +# Optional: +# +# --max_cov default: 200 +# +########################################### + +__EOUSAGE__ + + ; + +my $bam_file; +my $max_cov = 200; + +&GetOptions ( 'h' => \$help_flag, + 'bam=s' => \$bam_file, + 'max_cov=i' => \$max_cov + ); + + +if ($help_flag) { + die $usage; +} + +unless ($bam_file) { + die $usage; +} + +my $MAX_COV_DIST = 1e6; + +main: { + + my $curr_coord = 0; + my @cov_array; + my %retain; + my $prev_scaffold_name = ""; + + open(my $fh, "samtools view -h $bam_file | ") or die "Error, cannot view bam file: $bam_file"; + while(<$fh>) { + if (/^\@/) { + # header line + print $_; + next; + } + my $sam_line = $_; + my $sam_entry = new SAM_entry($sam_line); + my $scaffold_name = $sam_entry->get_scaffold_name(); + if ($scaffold_name ne $prev_scaffold_name) { + print STDERR "-processing $scaffold_name\n"; + ## reinit + $curr_coord = 0; + @cov_array = (); + %retain = (); + } + + my $mate_token; + my $mate_scaffold_pos; + if ($sam_entry->is_paired() && + $sam_entry->is_proper_pair() && + (! $sam_entry->is_mate_unmapped()) ) { + + $mate_scaffold_pos = $sam_entry->get_mate_scaffold_position(); + my $read_name = $sam_entry->get_core_read_name(); + $mate_token = join("$;", $read_name, $mate_scaffold_pos); + + } + + + my $aligned_pos = $sam_entry->get_aligned_position(); + if ($curr_coord < $aligned_pos) { + # shift it all over. + my $delta = $aligned_pos - $curr_coord; + if ($#cov_array > $delta) { + @cov_array = @cov_array[$delta .. $#cov_array]; + } + else { + @cov_array = (); + } + $curr_coord = $aligned_pos; + } + # fill the cov array: + my @align_coords = $sam_entry->get_alignment_coords(); + my $genome_align_coords = $align_coords[0]; + foreach my $coordset (@$genome_align_coords) { + &add_coverage(\@cov_array, $curr_coord, $coordset); + } + + my $pt_cov = $cov_array[0]; + + #print STDERR "$curr_coord\t$pt_cov\n"; + + if ($mate_token) { + if ($retain{$mate_token}) { + # force accept + delete $retain{$mate_token}; + print $sam_line; + next; + } + else { + # was the mate already seen and not retained? + if ($mate_scaffold_pos < $aligned_pos) { + next; + } + } + + } + + if ($pt_cov <= $max_cov) { + # selected. + print $sam_line; + if ($mate_token) { + # be sure to retain the mate too: + $retain{$mate_token} = 1; + } + } + + $prev_scaffold_name = $scaffold_name; + + } +} + +exit(0); + +#### +sub add_coverage { + my ($cov_array_aref, $start_coord, $coordset) = @_; + + my ($lend, $rend) = @$coordset; + $lend -= $start_coord; + $rend -= $start_coord; + + if ($rend > $MAX_COV_DIST) { + print STDERR "-exceeding max cov dist: start = $start_coord, lend=$lend, rend=$rend. Skipping...\n"; + return; + } + + for (my $i = $lend; $i <= $rend; $i++) { + $cov_array_aref->[$i]++; + } +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/ExitTester.jar b/99.scripts/trinity_utils/util/support_scripts/ExitTester.jar new file mode 100644 index 0000000..a75d6d8 Binary files /dev/null and b/99.scripts/trinity_utils/util/support_scripts/ExitTester.jar differ diff --git a/99.scripts/trinity_utils/util/support_scripts/GG_partitioned_trinity_aggregator.pl b/99.scripts/trinity_utils/util/support_scripts/GG_partitioned_trinity_aggregator.pl new file mode 100644 index 0000000..ce71959 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/GG_partitioned_trinity_aggregator.pl @@ -0,0 +1,34 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + + +my $usage = "usage: $0 token < trinity_fasta_files_listing\n\n"; + +my $token = $ARGV[0] or die $usage; + + +my $counter = 0; + +while () { + my $filename = $_; + chomp $filename; + unless (-e $filename) { + print STDERR "ERROR, filename: $filename is indicated to not exist.\n"; + next; + } + if (-s $filename) { + $counter++; + open (my $fh, $filename) or die "Error, cannot open file $filename"; + while (<$fh>) { + if (/>/) { + s/>/>${token}_${counter}\_/; + } + print; + } + } +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/support_scripts/SAM_coordSorted_fragment_Read_coverage_writer.pl b/99.scripts/trinity_utils/util/support_scripts/SAM_coordSorted_fragment_Read_coverage_writer.pl new file mode 100644 index 0000000..835fdf5 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/SAM_coordSorted_fragment_Read_coverage_writer.pl @@ -0,0 +1,110 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "usage: $0 coordSorted.file.sam\n\n"; + +my $sam_file = $ARGV[0] or die $usage; + + +main: { + + my %scaffold_to_coverage; # will retain all coverage information. + + my $sam_reader = new SAM_reader($sam_file); + + my $current_scaff = undef; + + my $counter = 0; + while ($sam_reader->has_next()) { + + $counter++; + if ($counter % 1000 == 0) { + print STDERR "\r[$counter] "; + } + + my $sam_entry = $sam_reader->get_next(); + + if ($sam_entry->is_query_unmapped()) { next; } + + + if (%scaffold_to_coverage && $sam_entry->get_scaffold_name() ne $current_scaff) { + &report_coverage(\%scaffold_to_coverage); + %scaffold_to_coverage = (); + } + + $current_scaff = $sam_entry->get_scaffold_name(); + + &add_coverage($sam_entry, \%scaffold_to_coverage); + + + } + + if (%scaffold_to_coverage) { + &report_coverage(\%scaffold_to_coverage); + } + + + exit(0); + +} + + +#### +sub report_coverage { + my ($scaffold_to_coverage_href) = @_; + + ## output the coverage information: + foreach my $scaffold (sort keys %$scaffold_to_coverage_href) { + + print "variableStep chrom=$scaffold\n"; + + my @coverage = @{$scaffold_to_coverage_href->{$scaffold}}; + + for (my $i = 1; $i <= $#coverage; $i++) { + my $cov = $coverage[$i] || 0; + + print "$i\t$cov\n"; + } + + } + + return; +} + + +#### +sub add_coverage { + my ($sam_entry, $scaffold_to_coverage_href) = @_; + + + ## If the entry is paired and the query alignment + ## is less than the paired position, then fill in the gap. + + my $scaffold = $sam_entry->get_scaffold_name(); + my $scaff_pos = $sam_entry->get_scaffold_position(); + + my @genome_coords; + + my ($genome_coords_aref, $query_coords_aref) = $sam_entry->get_alignment_coords(); + + foreach my $coordset (@$genome_coords_aref) { + my ($lend, $rend) = @$coordset; + + + ## add coverage: + for (my $i = $lend; $i <= $rend; $i++) { + $scaffold_to_coverage_href->{$scaffold}->[$i]++; + } + } + + + return; +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/SAM_coordSorted_fragment_coverage_writer2.pl b/99.scripts/trinity_utils/util/support_scripts/SAM_coordSorted_fragment_coverage_writer2.pl new file mode 100644 index 0000000..dc7134e --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/SAM_coordSorted_fragment_coverage_writer2.pl @@ -0,0 +1,161 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "\nusage: $0 coordSorted.file.sam [max_pair_distance=10000]\n\n"; + +my $sam_file = $ARGV[0] or die $usage; +my $MAX_PAIR_DIST = $ARGV[1] || 10000; + + +main: { + + my @coverage; + + my $sam_reader = new SAM_reader($sam_file); + + my $current_scaff = ""; + + my $counter = 0; + while ($sam_reader->has_next()) { + + $counter++; + if ($counter % 10000 == 0) { + print STDERR "\r[$counter] "; + } + + my $sam_entry = $sam_reader->get_next(); + + if ($sam_entry->get_scaffold_name() ne $current_scaff) { + &report_coverage(\@coverage) if @coverage; + @coverage = ([1,0]); # reinit + $current_scaff = $sam_entry->get_scaffold_name(); + print "variableStep chrom=$current_scaff\n"; + } + + &add_coverage($sam_entry, \@coverage); + + + } + + if (@coverage) { + &report_coverage(\@coverage); + } + + + exit(0); + +} + + +#### +sub report_coverage { + my ($coverage_aref) = @_; + + foreach my $cov_info (@$coverage_aref) { + my ($pos, $cov) = @$cov_info; + + print join("\t", $pos, $cov) . "\n"; + } + + return; +} + + + + + +#### +sub add_coverage { + my ($sam_entry, $coverage_aref) = @_; + + my $scaffold = $sam_entry->get_scaffold_name(); + my $scaff_pos = $sam_entry->get_scaffold_position(); + + if (@$coverage_aref) { + my $last_pos; + while (@$coverage_aref && $coverage_aref->[0]->[0] < $scaff_pos) { + my $pos_info = shift @$coverage_aref; + print join("\t", @$pos_info) . "\n"; + $last_pos = $pos_info->[0]; + } + + ## add any intervening positions to wig. + if ($last_pos) { + $last_pos++; + while ($last_pos < $scaff_pos) { + print "$last_pos\t0\n"; + $last_pos++; + } + } + } + + unless (@$coverage_aref) { + # prime it + push (@$coverage_aref, [$scaff_pos, 0]); + } + + + my @genome_coords; + + my ($genome_coords_aref, $query_coords_aref) = $sam_entry->get_alignment_coords(); + + foreach my $coordset (@$genome_coords_aref) { + my ($lend, $rend) = @$coordset; + push (@genome_coords, $lend, $rend); + } + + if ($sam_entry->is_paired() && $sam_entry->is_proper_pair()) { + + my $mate_scaffold = $sam_entry->get_mate_scaffold_name(); + my $mate_position = $sam_entry->get_mate_scaffold_position(); + + + if ( ($mate_scaffold eq $scaffold || $mate_scaffold eq '=') + && + $mate_position > $scaff_pos + && + $mate_position - $scaff_pos <= $MAX_PAIR_DIST) { + + push (@genome_coords, $mate_position-1); + } + } + + @genome_coords = sort {$a<=>$b} @genome_coords; + + my $lend = shift @genome_coords; + my $rend = pop @genome_coords; + + + if ($rend - $lend + 1 < $MAX_PAIR_DIST) { ## same deal for long introns. + + my $current_first_pos = $coverage_aref->[0]->[0]; + my $current_last_pos = $coverage_aref->[$#$coverage_aref]->[0]; + + #print "Current first pos: $current_first_pos\ncurrent last: $current_last_pos\n"; + + ## add positions for currently missing entries. + for (my $i = $current_last_pos + 1; $i <= $rend; $i++) { + push (@$coverage_aref, [$i, 0]); + } + + my $first_index = $lend - $current_first_pos; + my $last_index = $rend - $current_first_pos; + + ## add coverage: + for (my $i = $first_index; $i <= $last_index; $i++) { + + $coverage_aref->[$i]->[1]++; + + } + } + + return; +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/SAM_extract_properly_mapped_pairs.pl b/99.scripts/trinity_utils/util/support_scripts/SAM_extract_properly_mapped_pairs.pl new file mode 100644 index 0000000..7350e0f --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/SAM_extract_properly_mapped_pairs.pl @@ -0,0 +1,41 @@ +#!/usr/bin/env perl + +use strict; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "usage: $0 file.sam\n\n"; + +my $sam_file = $ARGV[0] or die $usage; + + + +main: { + + my $sam_reader = new SAM_reader($sam_file); + + + my $filtered_count = 0; + my $total_count = 0; + + + while ($sam_reader->has_next()) { + + my $sam_entry = $sam_reader->get_next(); + $total_count++; + + if (! $sam_entry->is_proper_pair()) { + $filtered_count++; + } + else { + print $sam_entry->toString() . "\n"; + } + } + + print STDERR "-filtered $filtered_count of $total_count indiv read alignments as not proper pairs = " . sprintf("%.2f", $filtered_count / $total_count * 100) . "\% of alignments\n"; + + exit(0); + +} diff --git a/99.scripts/trinity_utils/util/support_scripts/SAM_extract_uniquely_mapped_reads.pl b/99.scripts/trinity_utils/util/support_scripts/SAM_extract_uniquely_mapped_reads.pl new file mode 100644 index 0000000..2ab7e96 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/SAM_extract_uniquely_mapped_reads.pl @@ -0,0 +1,70 @@ +#!/usr/bin/env perl + +use strict; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "usage: $0 nameSorted.sam\n\n"; + +my $sam_file = $ARGV[0] or die $usage; + + + +main: { + + my $sam_reader = new SAM_reader($sam_file); + + + my $filtered_count = 0; + my $total_count = 0; + + my $prev_core_read_name = ""; + my @entries; + + while ($sam_reader->has_next()) { + + my $sam_entry = $sam_reader->get_next(); + + my $core_read_name = $sam_entry->get_core_read_name(); + if ($core_read_name ne $prev_core_read_name) { + &process_entries(@entries); + @entries = (); + } + + push (@entries, $sam_entry); + + $prev_core_read_name = $core_read_name; + } + + + if (@entries) { + &process_entries(@entries); + } + + + exit(0); + +} + + +#### +sub process_entries { + my @entries = @_; + + my %scaffolds; + foreach my $entry (@entries) { + my $scaff = $entry->get_scaffold_name(); + $scaffolds{$scaff}++; + } + + my $num_scaff = scalar(keys %scaffolds); + if ($num_scaff == 1) { + foreach my $entry (@entries) { + print $entry->toString() . "\n"; + } + } + + return; +} diff --git a/99.scripts/trinity_utils/util/support_scripts/SAM_filter_out_unmapped_reads.pl b/99.scripts/trinity_utils/util/support_scripts/SAM_filter_out_unmapped_reads.pl new file mode 100644 index 0000000..2091dee --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/SAM_filter_out_unmapped_reads.pl @@ -0,0 +1,41 @@ +#!/usr/bin/env perl + +use strict; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "usage: $0 file.sam\n\n"; + +my $sam_file = $ARGV[0] or die $usage; + + + +main: { + + my $sam_reader = new SAM_reader($sam_file); + + + my $filtered_count = 0; + my $total_count = 0; + + + while ($sam_reader->has_next()) { + + my $sam_entry = $sam_reader->get_next(); + $total_count++; + + if ($sam_entry->is_query_unmapped()) { + $filtered_count++; + } + else { + print $sam_entry->toString() . "\n"; + } + } + + print STDERR "-filtered $filtered_count of $total_count SAM entries as unaligned = " . sprintf("%.2f", $filtered_count / $total_count * 100) . "\% of SAM file corresponding to unmapped reads. Note, aligned reads may be counted multiple times due to multiple-mappings, so not the same as percent of unaligned reads.\n"; + + exit(0); + +} diff --git a/99.scripts/trinity_utils/util/support_scripts/SAM_ordered_pair_jaccard.pl b/99.scripts/trinity_utils/util/support_scripts/SAM_ordered_pair_jaccard.pl new file mode 100644 index 0000000..4836926 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/SAM_ordered_pair_jaccard.pl @@ -0,0 +1,126 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use FindBin; + + + +my $usage = <<_EOUSAGE_; + + +############################################################################################################## +# +# Required: +# +# --sam sam-formatted file +# +# Optional: +# +# --max_insert_size maximum insert size (default: 500) : used for pair coverage and jaccard +# --min_insert_size minimum insert size (default: 100) : used for jaccard +# +# --jaccard_win_length|-W default: 100 (requires --jaccard) +# +# --pseudocounts default: 1 +# -e write extended format, including 'single' and 'both' raw counts +# +############################################################################################################## + + + +_EOUSAGE_ + + ; + + +my $help_flag; + +# required +my $sam_file; + +# optional +my $max_insert_size = 500; +my $min_insert_size = 100; + +my $jaccard_win_length = 100; + +my $pseudocounts = 1; +my $extended_flag = 0; + + +&GetOptions ( 'h' => \$help_flag, + + # required + 'sam=s' => \$sam_file, + + # optional + 'max_insert_size=i' => \$max_insert_size, + 'min_insert_size=i' => \$min_insert_size, + + 'jaccard_win_length|W=i' => \$jaccard_win_length, + + + 'pseudocounts=i' => \$pseudocounts, + + 'e' => \$extended_flag, + + ); + + +if ($help_flag) { + die $usage; +} + + +unless ($sam_file) { + die $usage; +} + +if (@ARGV) { + die $usage; +} + + +my $util_dir = "$FindBin::RealBin"; + +main: { + + + ## generate the paired coverage info: + my $cmd = "$util_dir/SAM_to_frag_coords.pl --sam $sam_file " + . "--max_insert_size $max_insert_size " + . "--min_insert_size $min_insert_size "; ## writes file: $sam_file.frag_coords + + + &process_cmd($cmd) unless (-s "$sam_file.frag_coords"); + + ## compute jaccard coeff info for pair-support + $cmd = "$util_dir/ordered_fragment_coords_to_jaccard.pl --lend_sorted_frags $sam_file.frag_coords -W $jaccard_win_length " + . " --pseudocounts $pseudocounts "; + if ($extended_flag) { + $cmd .= " -e "; + } + + &process_cmd($cmd); + + + exit(0); + +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} diff --git a/99.scripts/trinity_utils/util/support_scripts/SAM_set_transcribed_orient_info.pl b/99.scripts/trinity_utils/util/support_scripts/SAM_set_transcribed_orient_info.pl new file mode 100644 index 0000000..87014ca --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/SAM_set_transcribed_orient_info.pl @@ -0,0 +1,104 @@ +#!/usr/bin/env perl + +use strict; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "usage: $0 file.sam SS_lib_type={F,R,FR,RF}\n\n"; + +my $sam_file = $ARGV[0] or die $usage; +my $SS_lib_type = $ARGV[1] or die $usage; + +unless ($SS_lib_type =~ /^(F|R|FR|RF)$/) { + die "Error, SS_lib_type must be F, R, FR, or RF"; +} + + +main: { + + my $sam_reader = new SAM_reader($sam_file); + + + my $num_verified = 0; + my $num_conflicted = 0; + + my $counter = 0; + + while ($sam_reader->has_next()) { + + $counter++; + print STDERR "\r[$counter] " if $counter % 10000 == 0; + + #if ($counter % 300000 == 0) { last; } ## debugging + + my $sam_entry = $sam_reader->get_next(); + + my $aligned_strand = $sam_entry->get_query_strand(); + my $opposite_strand = ($aligned_strand eq '+') ? '-' : '+'; + + my $transcribed_strand = $aligned_strand; + + if ($sam_entry->is_paired()) { + + unless ($SS_lib_type =~ /^(FR|RF)$/) { + die "Error, SS_lib_type: $SS_lib_type is not compatible with paired reads"; + } + + + if ($sam_entry->is_first_in_pair()) { + + if ($SS_lib_type eq "RF") { + $transcribed_strand = $opposite_strand; + } + } + + else { + # second in pair + if ($SS_lib_type eq "FR") { + $transcribed_strand = $opposite_strand; + } + } + } + else { + ## Unpaired or Single Reads + if ($SS_lib_type eq "R") { + $transcribed_strand = $opposite_strand; + } + } + + my $sam_text = $sam_entry->toString(); + + my @x = split(/\t/, $sam_text); + + my $strand_flag = "XS:A:$transcribed_strand"; + + if (my ($XS_A) = grep { $_ =~ /^XS:A:/ } @x) { + if ($XS_A eq $strand_flag) { + $num_verified++; + } + else { + $num_conflicted++; + print STDERR "Conflicting strand info: existing = $XS_A, expected = $strand_flag\n"; + @x = grep { $_ !~ /^XS:A:/ } @x; + push (@x, $strand_flag); + } + + } + else { + push (@x, $strand_flag); + } + + + print join("\t", @x) . "\n"; + } + + if ($num_conflicted || $num_verified) { + print STDERR "Percent entries with conflicted transcribed strand assignments = " + . sprintf("%.2f", $num_conflicted / ($num_conflicted + $num_verified) * 100) . "\n"; + } + + exit(0); + +} diff --git a/99.scripts/trinity_utils/util/support_scripts/SAM_strand_separator.pl b/99.scripts/trinity_utils/util/support_scripts/SAM_strand_separator.pl new file mode 100644 index 0000000..5984f19 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/SAM_strand_separator.pl @@ -0,0 +1,151 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; +use File::Basename; + +my $usage = "usage: $0 alignments.sam SS_lib_type=F,R,FR,RF\n\n"; + +my $sam_file = $ARGV[0] or die $usage; +my $SS_lib_type = $ARGV[1] or die $usage; + +unless ($SS_lib_type =~ /^(F|R|FR|RF)$/) { + die $usage; +} + + + +main: { + + my $sam_reader = new SAM_reader($sam_file); + + my $sam_basename = basename($sam_file); + + + open (my $plus_ofh, ">$sam_basename.+.sam") or die "Error, cannot write to $sam_basename.+.sam"; + open (my $minus_ofh, ">$sam_basename.-.sam") or die "Error, cannot write to $sam_basename.-.sam"; + + while (my $sam_entry = $sam_reader->get_next()) { + + if ($sam_entry->is_query_unmapped()) { + next; + } + + my $transcribed_strand = &get_transcribed_strand($sam_entry); + + my $ofh = ($transcribed_strand eq '+') ? $plus_ofh : $minus_ofh; + + print $ofh $sam_entry->toString() . "\n"; + + } + + close $plus_ofh; + close $minus_ofh; + + exit(0); +} + + + + +#### +sub get_transcribed_strand { + my ($sam_entry) = @_; + + unless ($SS_lib_type) { + confess "Error, SS_lib_type required as a parameter, possible values: RF,FR,F,R " . $sam_entry->toString(); + } + + my $aligned_strand = $sam_entry->get_query_strand(); + my $opposite_strand = ($aligned_strand eq '+') ? '-' : '+'; + + my $transcribed_strand; + + if (! $sam_entry->is_paired()) { + + my $is_long_read = &get_long_read_status($sam_entry); + + if ($is_long_read) { + + my $ts = &get_long_read_transcribed_strand($sam_entry); + if (defined($ts)) { + if ($ts ne $aligned_strand) { + print STDERR "-warning, long read " . $sam_entry->get_read_name() . " aligned as ($aligned_strand) but splicing indicates ($ts)\n"; + } + return($ts); + } + else { + return($aligned_strand); # should already be oriented in the forward orientation with respect to input sequence. + } + + } + + ## UNPAIRED or SINGLE READS + unless ($SS_lib_type =~ /^(F|R)$/) { + confess "Error, cannot have $SS_lib_type library type with unpaired reads " . $sam_entry->toString(); + } + + $transcribed_strand = ($SS_lib_type eq "F") ? $aligned_strand : $opposite_strand; + } + else { + + + ## paired RNA-Seq reads: left fragment is on the 3' end revcomped, and right fragment is at the 5' end sense strand. + + + unless ($SS_lib_type =~ /^(FR|RF)$/) { + confess "Error, cannot have $SS_lib_type library type with paired reads " . $sam_entry->toString(); + } + + if ($sam_entry->is_first_in_pair()) { + $transcribed_strand = ($SS_lib_type eq "FR") ? $aligned_strand : $opposite_strand; + } + else { + # second pair + $transcribed_strand = ($SS_lib_type eq "FR") ? $opposite_strand : $aligned_strand; + } + } + + + return($transcribed_strand); + +} + + + +#### +sub get_long_read_status { + my ($sam_entry) = @_; + + my @fields = $sam_entry->get_fields(); + @fields = @fields[11..$#fields]; + + if (grep { /^RG:Z:PBLR$/ } @fields) { + return(1); + } + else { + return(0); + } +} + +#### +sub get_long_read_transcribed_strand { + my ($sam_entry) = @_; + + my @fields = $sam_entry->get_fields(); + @fields = @fields[11..$#fields]; + + if (my $TS_field = grep { /^ts:A:/ } @fields) { + my @pts = split(":", $TS_field); + return($pts[2]); + } + else { + return(undef); + } +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/SAM_to_frag_coords.pl b/99.scripts/trinity_utils/util/support_scripts/SAM_to_frag_coords.pl new file mode 100644 index 0000000..35fe0ed --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/SAM_to_frag_coords.pl @@ -0,0 +1,285 @@ +#!/usr/bin/env perl + +# NOTE FOR WEB GUIDE http://trinityrnaseq.sourceforge.net/genome_guided_trinity.html +# % samtools view gsnap_out/gsnap.coordSorted.bam > gsnap.coordSorted.sam +# change to run samtools view with -F4 or -F12 if you are plannign to use --no_single + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use SAM_reader; +use SAM_entry; +use COMMON; + +use Getopt::Long qw(:config no_ignore_case bundling); + +use Carp; +use Data::Dumper; +use Cwd; + +$ENV{LC_ALL} = 'C'; # critical for proper sorting using [system "sort -k1,1 ..."] within the perl script + + +my $usage = <<_EOUSAGE_; + +####################################################################### +# +# Required: +# +# --sam SAM-formatted file +# +# Optional: +# +# --no_single no single reads +# +# --min_insert_size minimum distance between proper pair's starting positions. (default: 100) +# --max_insert_size maximum distance between proper pair's starting positions. (default: 500) +# +# --sort_buffer default '10G', amount of RAM to allocate to sorting +# +# -d debug mode +# +####################################################################### + + +_EOUSAGE_ + + ; + +my $sam_file; +my $full_flag = 0; +my $MAX_INSERT_SIZE = 500; +my $MIN_INSERT_SIZE = 100; +my $DEBUG = 0; +my $help_flag = 0; +my $NO_SINGLE; +my $CPU = 1; +my $sort_buffer = '10G'; +&GetOptions( + 'h' => \$help_flag, + + 'sam=s' => \$sam_file, + 'CPU=i' => \$CPU, + 'sort_buffer=s' => \$sort_buffer, + 'no_single' => \$NO_SINGLE, + 'max_insert_size=i' => \$MAX_INSERT_SIZE, + 'min_insert_size=i' => \$MIN_INSERT_SIZE, + + 'd' => \$DEBUG, + + ); + + +if ($help_flag) { + die $usage; +} + + +unless ($sam_file) { + die $usage; +} + +my $sort_exec = &COMMON::get_sort_exec($CPU); + + +main: { + + + my @frags; + + my $read_coords_file = "$sam_file.read_coords"; + my $read_coords_checkpoint = "$read_coords_file.ok"; + + &extract_read_coords($read_coords_file) unless (-s $read_coords_checkpoint); + &process_cmd("touch $read_coords_checkpoint"); + + my $pair_coords_file = "$sam_file.frag_coords"; + &extract_frag_coords($read_coords_file, $pair_coords_file); + + + exit(0); + + + +} + +#### +sub extract_read_coords { + my ($read_coords_file) = @_; + # AP: I think this is very expensive (more than 20' for 12G SAM) + # option 1) the same information could be derived by parsing the BAM file + # and cut -f 1,2,3,7,8,4,9,10 -> there will be two identical (1) fields, if paired the one with (9)>0 is the /1 one||the one on the plus strand (from (2)) could be assigned as the /1 one + # generally, i don't see the need to have to create a SAM file in the first place... + # another benefit of samtools view bam is that we could extract each scaffold separately (would help with sort...) + # option 2) use samtools depth instead of wig file??? + + ## Every read processed. + + print STDERR "-extracting read coordinates from $sam_file into $read_coords_file\n\n"; + + my $sam_reader = new SAM_reader($sam_file); + + open (my $ofh, ">$read_coords_file") or die "Error, cannot write to file $read_coords_file"; + + while ($sam_reader->has_next()) { + + my $sam_entry = $sam_reader->get_next(); + + # unless ($sam_entry->get_mate_scaffold_name() eq "=" || $sam_entry->get_mate_scaffold_name() eq $sam_entry->get_scaffold_name()) { next; } + # commenting out above - just describe the reads, let the other routine handle the fragment definition. + + + eval { + my $scaffold = $sam_entry->get_scaffold_name(); + my $core_read_name = $sam_entry->get_core_read_name(); + my $read_name = $sam_entry->get_read_name(); + my $full_read_name = $sam_entry->reconstruct_full_read_name(); + + #print "read_name: $read_name, full_read_name: $full_read_name\n"; + + + my $pair_side = "."; + if ($full_read_name =~ m|/([12])$|) { + $pair_side = $1; + } + elsif ($read_name =~ m|/([12])$|) { + $pair_side = $1; + } + + my $mate_scaff_pos = $sam_entry->get_mate_scaffold_position(); + my ($read_start, $read_end) = $sam_entry->get_genome_span(); + + + print $ofh join("\t", $scaffold, $core_read_name, $pair_side, $read_start, $read_end) . "\n" if $scaffold ne '*' && $read_start && $read_end; + }; + + if ($@) { + print STDERR "******\nError parsing SAM entry: " . Dumper($sam_entry) . " \n$@\n******\n\n\n"; + } + } + + close $ofh; + + + return; + +} + + +#### +sub extract_frag_coords { + my ($read_coords_file, $pair_frag_coords_file) = @_; + unless (-s "$read_coords_file.sort_by_readname" ){ + ## sort by scaffold, then by read name + my $cmd = "$sort_exec -S$sort_buffer -T . -k1,1 -k2,2 -k4,4n $read_coords_file > $read_coords_file.sort_by_readname"; + &process_cmd($cmd); + rename("$read_coords_file.sort_by_readname", $read_coords_file); + # so that sort is not re-done... + #symlink(&create_full_path($read_coords_file),&create_full_path("$read_coords_file.sort_by_readname")); + &process_cmd("cp " . &create_full_path($read_coords_file) . " " . &create_full_path("$read_coords_file.sort_by_readname")); # some systems dont like symlinks + } + + ## define fragment pair coordinate span + open (my $fh, "$read_coords_file") or die $!; + + open (my $ofh, ">$pair_frag_coords_file") or die $!; + + my $prev_reported_pair = ""; + my $prev_reported_single = ""; + + my $first = <$fh>; + chomp $first; + while (my $second = <$fh>) { + next if ($first =~/^\*/ && $second =~/^\*/); # both unmapped + chomp $second; + + my ($scaffA, $readA, $readA_pair_side, $lendA, $rendA) = split(/\t/, $first); + my ($scaffB, $readB, $readB_pair_side, $lendB, $rendB) = split(/\t/, $second); + + my $got_pair_flag = 0; + if ($readA eq $readB + && $scaffA eq $scaffB + && $readA_pair_side ne $readB_pair_side + && $readA_pair_side =~ /\d/ && $readB_pair_side =~ /\d/) { + + my @coords = sort {$a<=>$b} ($lendA, $rendA, $lendB, $rendB); + my $min = shift @coords; + my $max = pop @coords; + + my $insert_size = $max - $min + 1; + if ($insert_size >= $MIN_INSERT_SIZE && $insert_size <= $MAX_INSERT_SIZE && $readA ne $prev_reported_pair) { + # treat as proper pair + + print $ofh join("\t", $scaffA, $readA, $min, $max) . "\n"; + $prev_reported_pair = $readA; # only one proper pair to be reported. - not random though, first one encountered. + + } + else { + # treat as unpaired reads + unless ($NO_SINGLE) { + if ($prev_reported_single ne $readA) { + print $ofh join("\t", $scaffA, $readA . "/$readA_pair_side", $lendA, $rendA) . "\n"; + print $ofh join("\t", $scaffB, $readB . "/$readB_pair_side", $lendB, $rendB) . "\n"; + } + $prev_reported_single = $readA; + } + } + + $first = <$fh>; # prime first + chomp $first if $first; + } + else { + # not paired + unless ($NO_SINGLE) { + if ($prev_reported_single ne $readA) { + print $ofh join("\t", $scaffA, $readA . "/$readA_pair_side", $lendA, $rendA) . "\n"; + } + $prev_reported_single = $readA; # only letting one slip through. + } + + $first = $second; + next; + } + + + + } + + close $ofh; + close $fh; + + + my $cmd = "$sort_exec -S$sort_buffer -T . -k1,1 -k3,3n $pair_frag_coords_file > $pair_frag_coords_file.coord_sorted"; + &process_cmd($cmd) unless -s "$pair_frag_coords_file.coord_sorted"; + + rename("$pair_frag_coords_file.coord_sorted", $pair_frag_coords_file); + + return; +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + +#### +sub create_full_path { + my ($path) = @_; + + unless ($path =~ /^\//) { + $path = cwd() . "/$path"; + } + + return($path); +} diff --git a/99.scripts/trinity_utils/util/support_scripts/add_LR_reads_to_iworm_bundle.pl b/99.scripts/trinity_utils/util/support_scripts/add_LR_reads_to_iworm_bundle.pl new file mode 100644 index 0000000..2b8b48f --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/add_LR_reads_to_iworm_bundle.pl @@ -0,0 +1,55 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "\n\n\tusage: $0 iworm_bundles_fasta_file sorted_reads_to_components_file\n\n\n"; + +my $iworm_bundles_fasta_file = $ARGV[0] or die $usage; +my $reads_to_components_file = $ARGV[1] or die $usage; + + +main: { + + my %component_to_LRs; + { + open (my $fh, $reads_to_components_file) or die "Error, cannot open file: $reads_to_components_file"; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my $component_id = $x[0]; + my $read_name = $x[1]; + + if ($read_name =~ /^>LR\$\|/) { + # got a long read! + my $seq = $x[3]; + $component_to_LRs{$component_id} .= "X$seq"; + } + } + close $fh; + } + + ## now add them to the compoennts: + open (my $fh, $iworm_bundles_fasta_file) or die "Error, cannot open file: $iworm_bundles_fasta_file"; + while (my $header = <$fh>) { + + chomp $header; + + my $seq = <$fh>; + + chomp $seq; + + $header =~ /^>s_(\d+)/ or die "Error, cannot decipher component id"; + my $component_id = $1; + + if (my $LRs = $component_to_LRs{$component_id}) { + $seq .= $LRs; + } + + print "$header\n$seq\n"; + } + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/support_scripts/annotate_chrysalis_welds_with_iworm_names.pl b/99.scripts/trinity_utils/util/support_scripts/annotate_chrysalis_welds_with_iworm_names.pl new file mode 100644 index 0000000..2fffa05 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/annotate_chrysalis_welds_with_iworm_names.pl @@ -0,0 +1,48 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "\n\n\tusage: $0 iworm.fasta iworm_cluster_welds_graph.txt\n\n"; + +my $iworm_fa = $ARGV[0] or die $usage; +my $iworm_cluster_welds_file = $ARGV[1] or die $usage; + + main: { + + + my %counter_to_iworm; + + { + my $counter = 0; + open(my $fh, $iworm_fa) or die "Error, cannot open $iworm_fa file"; + while (<$fh>) { + if (/^>(\S+)/) { + $counter_to_iworm{$counter} = $1; + $counter++; + } + } + close $fh; + } + + open(my $fh, $iworm_cluster_welds_file) or die "Error, cannot open file: $iworm_cluster_welds_file"; + while(<$fh>) { + chomp; + my ($idx_A, $ptr, $idx_B, @rest) = split(/\s+/); + my $iworm_A = $counter_to_iworm{$idx_A}; + + if (! defined $iworm_A) { die "Error, no iworm acc for index: $idx_A"; } + + my $iworm_B = $counter_to_iworm{$idx_B}; + + if (! defined $iworm_B) { die "Error, no iworm acc for index: $idx_B"; } + + print join(" ", $idx_A, $iworm_A, $ptr, $idx_B, $iworm_B, @rest) . "\n"; + } + close $fh; + + +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/support_scripts/batch_cmds.pl b/99.scripts/trinity_utils/util/support_scripts/batch_cmds.pl new file mode 100644 index 0000000..a18d92f --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/batch_cmds.pl @@ -0,0 +1,91 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use Getopt::Long qw(:config posix_default no_ignore_case bundling pass_through); +use List::Util qw(shuffle); + + +my $usage = <<__EOUSAGE__; + +###################################################################################### +# +# Required: +# +# --cmds cmds file +# +# --max_batch_size maximum batch size +# +# Optional: +# +# --shuffle shuffle the commands in random order before batching +# +####################################################################################### + + +__EOUSAGE__ + + ; + + +my $help_flag; + +my $cmds_file; +my $max_batch_size; +my $shuffle_flag = 0; + +&GetOptions ( 'h' => \$help_flag, + 'cmds=s' => \$cmds_file, + 'max_batch_size=i' => \$max_batch_size, + 'shuffle' => \$shuffle_flag); + + +if ($help_flag) { + die $usage; +} + +unless ($cmds_file && $max_batch_size) { + die $usage; +} + + +main: { + + my @cmds = `cat $cmds_file`; + chomp @cmds; + + if ($shuffle_flag) { + @cmds = shuffle(@cmds); + } + + my $num_cmds = scalar(@cmds); + + my $cmds_per_batch = int($num_cmds / $max_batch_size); + if ($cmds_per_batch < 1) { + $cmds_per_batch = 1; + } + + while (@cmds) { + + my @batch; + for (1..$cmds_per_batch) { + my $cmd = shift @cmds; + if ($cmd) { + push (@batch, $cmd); + } + } + my $batched_cmds = join(" && ", @batch); + + print "$batched_cmds\n"; + } + + exit(0); +} + + + + + + + diff --git a/99.scripts/trinity_utils/util/support_scripts/bowtie2_wrapper.pl b/99.scripts/trinity_utils/util/support_scripts/bowtie2_wrapper.pl new file mode 100644 index 0000000..e668b90 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/bowtie2_wrapper.pl @@ -0,0 +1,338 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use Cwd; +use File::Basename; +use Carp; +use Data::Dumper; + +use Getopt::Long qw(:config no_ignore_case bundling); + +$ENV{PATH} .= "\:$FindBin::RealBin/../trinity-plugins/rsem/sam/"; # include samtools in path, already included in rsem build. + +$ENV{LC_ALL} = 'C'; # critical for proper sorting using [system "sort -k1,1 ..."] within the perl script + + +my $output_prefix = "bowtie2"; + + +my $usage = <<_EOUSAGE_; + +################################################################################################################ +# +# --left and --right reads +# +# or +# +# --single reads +# +# Required inputs: +# +# --target multi-fasta file containing the target sequences (should be named {refName}.fa ) +# +# --seqType fa | fq (fastA or fastQ format) +# +# +# Optional: +# +# --SS_lib_type strand-specific library type: single: F or R paired: FR or RF +# examples: single RNA-Ligation method: F +# single dUTP method: R +# paired dUTP method: RF +# +# --num_top_hits (default: 20) +# +# --retain_intermediate_files retain all the intermediate sam files produced (they take up lots of space! and there's lots of them) +# +# --max_dist_between_pairs default (2000) +# +# --just_prep_build just prepare the bowtie-build and stop. +# +# --output_prefix|o prefix for output filename (default: $output_prefix) +# +# --CPU number of threads +# +# ## General options +# +# Any options after '--' are passed onward to the alignments programs (except BLAT -which has certain options exposed above). +# For example, to set the number of processors used by Bowtie to 16 then use: +# -- -p 16 +# +#################################################################################################################### +# +# Example commands: +# +# $0 --seqType fq --left left.fq --right right.fq --target Trinity.fasta +# +# +#################################################################################################################### + + + +_EOUSAGE_ + + ; + + +my $help_flag; +my $target_db; +my $left_file; +my $right_file; +my $single_file; + +my $num_top_hits = 20; +my $max_dist_between_pairs = 2000; +my $seqType; +my $SS_lib_type; +my $retain_intermediate_files_flag = 0; + +my $trinity_mode; +my $CPU = 1; + +my $JUST_PREP_BUILD = 0; + +unless (@ARGV) { + die $usage; +} + + +&GetOptions ( 'h' => \$help_flag, + + ## required inputs + 'left=s' => \$left_file, + 'right=s' => \$right_file, + 'single=s' => \$single_file, + + "target=s" => \$target_db, + "seqType=s" => \$seqType, + + "CPU=i" => \$CPU, + + + ## Optional: + "SS_lib_type=s" => \$SS_lib_type, + + 'output_prefix|o=s' => \$output_prefix, + + 'num_top_hits=i' => \$num_top_hits, + + 'max_dist_between_pairs=i' => \$max_dist_between_pairs, + + 'retain_intermediate_files' => \$retain_intermediate_files_flag, + + 'just_prep_build' => \$JUST_PREP_BUILD, + + ); + + +if ($help_flag) { die $usage; } + + +unless ($target_db && -s $target_db) { + die $usage . "Must specify target_db and it must exist at that location"; +} + +unless ($JUST_PREP_BUILD || ($seqType && $seqType =~ /^(fq|fa)$/)) { + die $usage . ", sorry do not understand seqType $seqType"; +} + + +unless ($JUST_PREP_BUILD + || + ($left_file && $right_file) + || + ($single_file) + + ) { + die $usage . "sorry, cannot find files $left_file $right_file $single_file"; +} + + +if ($SS_lib_type && $SS_lib_type !~ /^(F|R|FR|RF)$/) { + die "Error, SS_lib_type must be one of the following: (F, R, FR, RF) "; +} + + + +## check for required programs +{ + + my @required_progs = qw(samtools bowtie2-build bowtie2); + + foreach my $prog (@required_progs) { + my $path = `sh -c "command -v $prog"`; + unless ($path =~ /^\//) { + die "Error, path to required $prog cannot be found"; + } + } +} + + +my $util_dir = "$FindBin::RealBin"; + +my ($start_dir, $work_dir, $num_hits); + + +main: { + $start_dir = cwd(); + + $left_file = &build_full_paths($left_file, $start_dir) if $left_file; + $right_file = &build_full_paths($right_file, $start_dir) if $right_file; + $target_db = &build_full_paths($target_db, $start_dir); + + $single_file = &build_full_paths($single_file, $start_dir) if $single_file; + + unless (-s "$target_db.fai") { + &process_cmd("samtools faidx $target_db"); + } + + + ###################################### + ## Prep the bowtie index of the target + ###################################### + + my $index_ext = "bt2"; + + my @bowtie_build_files = <$target_db.*.$index_ext>; + print STDERR "bt2 index files: " . Dumper(\@bowtie_build_files); + my $index_file_checkpoint = "$target_db._${index_ext}_idx_.ok"; + unless (@bowtie_build_files && -e $index_file_checkpoint) { + + print STDERR "Note - bowtie-build indices do not yet exist. Indexing genome now.\n"; + ## run bowtie-build: + my $builder = "bowtie2-build"; + + my $cmd = "$builder -q $target_db $target_db"; + &process_cmd($cmd); + + &process_cmd("touch $index_file_checkpoint"); + + } + + if ($JUST_PREP_BUILD) { + print STDERR "Just preparing build, stopping now.\n"; + exit(0); + } + + my @entries; + + my $format = ($seqType eq "fq") ? "-q" : "-f"; + + my $output_file = "$output_prefix.coordSorted.bam"; + + my $cmd; + if ($left_file && $right_file) { + + $cmd = "bash -c \"set -o pipefail; bowtie2 --local -k $num_top_hits --threads $CPU --no-unal $format -X $max_dist_between_pairs -x $target_db -1 $left_file -2 $right_file | samtools sort -@ $CPU -o - - > $output_file\" "; + + + } + else { + + $cmd = "bash -c \"set -o pipefail; bowtie2 --local -k $num_top_hits --threads $CPU --no-unal $format -x $target_db -U $single_file | samtools sort -o - - > $output_file\" "; + + } + + &process_cmd($cmd) unless (-e "$output_file.ok"); + + &process_cmd("touch $output_file.ok") unless (-e "$output_file.ok"); + + + my @to_delete; + + if ($SS_lib_type) { + + ## strand-specific + ## separate the sam based on strand, and create separate bam files. (for convenience sake) + + $cmd = "$util_dir/SAM_strand_separator.pl $output_file $SS_lib_type"; + &process_cmd($cmd); + + + foreach my $sam_file ("$output_file.+.sam", "$output_file.-.sam") { + + my $bam_file = $sam_file; + $bam_file =~ s/\.sam$/\.bam/; # add suffix below + + unless (-s $sam_file) { next; } # empty file + + push (@to_delete, $sam_file); + + $cmd = "samtools view -bt $target_db.fai $sam_file | samtools sort -o - - > $bam_file"; # .bam ext added auto + + &process_cmd($cmd); + + $cmd = "samtools index $bam_file"; + &process_cmd($cmd); + + } + } + + + ## do final cleanpup of intermediate files + unless ($retain_intermediate_files_flag) { + &purge_files(@to_delete); + } + @to_delete = (); + + + + exit(0); +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + + if ($ret) { + confess "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + + +#### +sub build_full_paths { + my ($path, $start_dir) = @_; + + my @paths; + + foreach my $p (split(/,/, $path)) { + + if ($p !~ /^\//) { + $p = "$start_dir/$p"; + } + push (@paths, $p); + } + + $path = join(",", @paths); + + return($path); +} + +#### +sub purge_files { + my @to_delete = @_; + + foreach my $file (@to_delete) { + if (-e $file || -l $file) { + print STDERR "-cleaning up and removing intermediate file: $file\n"; + unlink($file); + } + else { + #print STDERR "-warning: cannot locate file $file targeted for deletion.\n"; + } + } + + return; +} diff --git a/99.scripts/trinity_utils/util/support_scripts/define_SAM_coverage_partitions2.pl b/99.scripts/trinity_utils/util/support_scripts/define_SAM_coverage_partitions2.pl new file mode 100644 index 0000000..e2bfafd --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/define_SAM_coverage_partitions2.pl @@ -0,0 +1,74 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use WigParser; + +my $usage = "usage: $0 strand_coverage.wig strand[+-]\n\n"; + +# need strand value to include in the GFF file. + +my $wig_file = $ARGV[0] or die $usage; +my $strand = $ARGV[1] or die $usage; + +main: { + + my $scaffold = ""; + my $lend = undef; + my $rend = undef; + + open (my $fh, $wig_file) or die "Error, cannot open file $wig_file"; + while (<$fh>) { + chomp; + if (/variableStep chrom=(\S+)/) { + if ($lend && $rend) { + print "$scaffold\tpartition\tregion\t$lend\t$rend\t.\t$strand\t.\t.\n"; + $lend = undef; + $rend = undef; + } + + $scaffold = $1; + next; + } + + my @x = split(/\t/); + my ($pos, $cov) = @x; + + if ($cov) { + + if (! defined $lend) { + $lend = $pos; + $rend = $pos; + } + else { + $rend = $pos; + } + } + else { + # no coverage... may have ended a block. + if ($lend) { + # report coverage block: + my $len = $rend - $lend + 1; + print "$scaffold\tpartition\tregion\t$lend\t$rend\t.\t$strand\t.\t.\t$len\n"; + + # reinit + $lend = undef; + $rend = undef; + } + + } + } + + + # get last block + if ($lend && $rend) { + print "$scaffold\tpartition\tregion\t$lend\t$rend\t.\t$strand\t.\t.\n"; + } + + exit(0); + +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/define_coverage_partitions.pl b/99.scripts/trinity_utils/util/support_scripts/define_coverage_partitions.pl new file mode 100644 index 0000000..2d01b86 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/define_coverage_partitions.pl @@ -0,0 +1,78 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use WigParser; + +my $usage = "usage: $0 strand_coverage.wig min_coverage strand[+-]\n\n"; + +# need strand value to include in the GFF file. + +my $wig_file = $ARGV[0] or die $usage; +my $min_coverage = $ARGV[1] or die $usage; +my $strand = $ARGV[2] or die $usage; + +main: { + + my $scaffold = ""; + my $lend = undef; + my $rend = undef; + + open (my $fh, $wig_file) or die "Error, cannot open file $wig_file"; + while (<$fh>) { + unless (/\w/) { next; } + chomp; + if (/variableStep chrom=(\S+)/) { + if ($lend && $rend) { + my $len = $rend - $lend + 1; + print "$scaffold\tpartition\tregion\t$lend\t$rend\t.\t$strand\t.\t.\t$len\n"; + $lend = undef; + $rend = undef; + } + + $scaffold = $1; + next; + } + + my @x = split(/\t/); + my ($pos, $cov) = @x; + + if ($cov >= $min_coverage) { + + if (! defined $lend) { + $lend = $pos; + $rend = $pos; + } + else { + $rend = $pos; + } + } + else { + # no coverage... may have ended a block. + if ($lend) { + # report coverage block: + my $len = $rend - $lend + 1; + print "$scaffold\tpartition\tregion\t$lend\t$rend\t.\t$strand\t.\t.\t$len\n"; + + # reinit + $lend = undef; + $rend = undef; + } + + } + } + + + # get last block + if ($lend && $rend) { + my $len = $rend - $lend + 1; + print "$scaffold\tpartition\tregion\t$lend\t$rend\t.\t$strand\t.\t.\t$len\n"; + } + + exit(0); + +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/eXpress_trans_to_gene_results.pl b/99.scripts/trinity_utils/util/support_scripts/eXpress_trans_to_gene_results.pl new file mode 100644 index 0000000..5b250c1 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/eXpress_trans_to_gene_results.pl @@ -0,0 +1,161 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Data::Dumper; + +my $usage = "\n\nusage: $0 results.xprs gene_to_trans_map_file.txt\n\n\n"; + +my $results_xprs = $ARGV[0] or die $usage; +my $gene_to_trans_map_file = $ARGV[1] or die $usage; + + +main: { + + my %trans_to_gene_info; + { + open (my $fh, $gene_to_trans_map_file) or die "Error, cannot open file $gene_to_trans_map_file"; + while (<$fh>) { + unless (/\w/) { next; } + chomp; + my ($gene, $trans, @rest) = split(/\s+/); + unless ($gene && $trans) { + die "Error, cannot extract gene & trans relationship from line $_ of file $gene_to_trans_map_file"; + } + $trans_to_gene_info{$trans} = $gene; + } + close $fh; + } + + + open (my $fh, $results_xprs) or die "Error, cannot open file $results_xprs"; + my $header = <$fh>; + chomp $header; + my %field_index; + my @fields = split(/\t/, $header); + { + + for (my $i = 1; $i <= $#fields; $i++) { + my $field = $fields[$i]; + $field_index{$field} = $i; + } + } + + + my %gene_data; + while (<$fh>) { + chomp; + my @x = split(/\t/); + + my $trans_id = $x[ $field_index{target_id} ]; + my $fpkm = $x[ $field_index{fpkm} ]; + my $length = $x[ $field_index{length} ]; + my $eff_length = $x[ $field_index{eff_length} ]; + my $eff_counts = $x[ $field_index{eff_counts} ]; + my $tpm = $x[ $field_index{tpm} ]; + + my $gene = $trans_to_gene_info{$trans_id} or die "Error, cannot find gene identifier for transcript [$trans_id] "; + + push (@{$gene_data{$gene}}, { trans_id => $trans_id, + fpkm => $fpkm, + length => $length, + eff_length => $eff_length, + eff_counts => $eff_counts, + tpm => $tpm, + }); + + + } + close $fh; + + + ## Output gene summaries: + + print $header . "\n"; + + foreach my $gene (keys %gene_data) { + my @trans_structs = @{$gene_data{$gene}}; + + my @trans_ids; + my $sum_counts = 0; + my $sum_fpkm = 0; + my $sum_tpm = 0; + + my $counts_per_len_sum = 0; + my $counts_per_eff_len_sum = 0; + + my $sum_lengths = 0; + my $sum_eff_lengths = 0; + + my $num_trans = scalar(@trans_structs); + + foreach my $struct (@trans_structs) { + + #print Dumper($struct); + + my $trans_id = $struct->{trans_id}; + my $fpkm = $struct->{fpkm}; + my $tpm = $struct->{tpm}; + my $length = $struct->{length}; + + + my $eff_length = $struct->{eff_length}; + my $eff_counts = $struct->{eff_counts}; + + unless ($eff_length > 0) { + $eff_length = 1; # cannot have zero length feature! + } + + unless ($length > 0 && $eff_length > 0) { + die "Error, length: $length, eff_length: $eff_length" . Dumper($struct); + } + + $sum_lengths += $length; + $sum_eff_lengths += $eff_length; + + $counts_per_len_sum += $eff_counts/$length; + + $counts_per_eff_len_sum += $eff_counts/$eff_length; + + $sum_counts += $eff_counts; + $sum_fpkm += $fpkm; + + $sum_tpm += $tpm; + } + + my $gene_length = $sum_lengths / $num_trans; + my $gene_eff_length = $sum_eff_lengths / $num_trans; + if ($sum_counts) { + # compute it based on the formula: FPKM_gene = FPKM_isoA + FPKM_IsoB + ... + eval { + $gene_length = $sum_counts / $counts_per_len_sum; + $gene_eff_length = $sum_counts / $counts_per_eff_len_sum; + }; + if ($@) { + print STDERR "$@\n" . Dumper(\@trans_structs); + die; + } + } + + my %gene_info = ( target_id => $gene, + fpkm => sprintf("%.2f", $sum_fpkm), + length => sprintf("%.2f", $gene_length), + eff_length => sprintf("%.2f", $gene_eff_length), + eff_counts => sprintf("%.2f", $sum_counts), + tpm => $sum_tpm, + ); + + my @vals; + foreach my $field (@fields) { + my $result = $gene_info{$field}; + unless (defined $result) { + $result = "NA"; + } + push (@vals, $result); + } + print join("\t", @vals) . "\n"; + } + + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/support_scripts/ensure_coord_sorted_sam.pl b/99.scripts/trinity_utils/util/support_scripts/ensure_coord_sorted_sam.pl new file mode 100644 index 0000000..a258400 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/ensure_coord_sorted_sam.pl @@ -0,0 +1,80 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "\n\n\tusage: $0 coordinate_sorted.bam|sam\n\n"; + +my $bam_file = $ARGV[0] or die $usage; + + + +my $num_records_to_validate = 1000; + +main: { + + my $order_counter = 0; + + my $sam_reader = new SAM_reader($bam_file); + + + my $ordered_record_counter = 0; + + my $prev_record = $sam_reader->get_next(); + unless ($prev_record) { + print STDERR "WARNING, bam file appears to not have any reads! exiting gracefully.\n"; + exit(0); + } + + + my %scaff_seen; + $scaff_seen{ $prev_record->get_scaffold_name() } = 1; + + while (my $sam_entry = $sam_reader->get_next()) { + + + if ($prev_record->get_scaffold_name() eq $sam_entry->get_scaffold_name()) { + ## ensure coordinates are in order + + if ($sam_entry->get_scaffold_position() < $prev_record->get_scaffold_position()) { + + die "Error, read entries are out of order:\n" + . $prev_record->get_original_line() . "\n" + . $sam_entry->get_original_line() . "\n" + . "\n\nBe sure to use a coordinate-sorted bam file\n"; + } + elsif ($sam_entry->get_scaffold_position() > $prev_record->get_scaffold_position()) { + # good, as we expect + $ordered_record_counter++; + + if ($ordered_record_counter >= $num_records_to_validate) { + print STDERR "-appears to be a coordinate sorted bam file. ok.\n"; + exit(0); + } + } + } + elsif ($scaff_seen{ $sam_entry->get_scaffold_name() } ) { + die "Error, bam file doesn't appear to be coordinate sorted. Scaffold: " . $sam_entry->get_scaffold_name() . " is out of order. "; + } + + $scaff_seen{ $sam_entry->get_scaffold_name() } = 1; + + $prev_record = $sam_entry; + } + + + print STDERR "Warning: didn't find at least $num_records_to_validate BAM records properly ordered along a single scaffold... either the file contains few reads per scaffold or there may be a problem.\n"; + + exit(0); + + +} + + + + diff --git a/99.scripts/trinity_utils/util/support_scripts/extract_reads_per_partition.pl b/99.scripts/trinity_utils/util/support_scripts/extract_reads_per_partition.pl new file mode 100644 index 0000000..8717687 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/extract_reads_per_partition.pl @@ -0,0 +1,308 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use File::Path; +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); +use File::Basename; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use Nuc_translator; +use SAM_reader; +use SAM_entry; + + +my $usage = <<__EOUSAGE__; + +################################################################ +# +# Required: +# +# --partitions_gff list of partitions in gff format. +# --coord_sorted_SAM coordinate-sorted SAM file +# +# Options: +# +# --SS_lib_type [SS_lib_type=F,R,FR,RF] +# +# --parts_per_directory default: 100 +# --min_reads_per_partition default: 10 +# +################################################################# + +__EOUSAGE__ + + ; + + +my $partitions; +my $alignments_sam; +my $SS_lib_type; + +my $PARTS_PER_DIR = 100; + +my $MIN_READS_PER_PARTITION = 10; + + +&GetOptions ( + 'partitions_gff=s' => \$partitions, + 'coord_sorted_SAM=s' => \$alignments_sam, + 'SS_lib_type=s' => \$SS_lib_type, + 'parts_per_directory=i' => \$PARTS_PER_DIR, + 'min_reads_per_partition=i' => \$MIN_READS_PER_PARTITION, + ); + +unless ($partitions && $alignments_sam) { + die $usage; +} + + +main: { + + + my $partitions_dir = "Dir_". basename($partitions); + unless (-d $partitions_dir) { + mkdir ($partitions_dir) or die "Error, cannot mkdir $partitions_dir"; + } + open (my $track_fh, ">$partitions_dir.listing") or die $!; + + my %scaff_to_partitions = &parse_partitions($partitions); + + my @ordered_partitions; + my $current_partition = undef; + + my $ofh; + + my $sam_ofh; + + my $partition_counter = 0; + + my $part_file = ""; + my $sam_part_file = ""; + my $read_counter = 0; + + my $current_scaff = ""; + + my $sam_reader = new SAM_reader($alignments_sam); + while (my $sam_entry = $sam_reader->get_next()) { + + my $acc = $sam_entry->reconstruct_full_read_name(); + my $scaff = $sam_entry->get_scaffold_name(); + next if $scaff eq '*'; + + my $seq = $sam_entry->get_sequence(); + next if $seq eq "*"; + + my $start = $sam_entry->get_aligned_position(); + + my $read_name = $sam_entry->get_read_name(); # raw from sam file + if ($acc !~ /\/[12]$/ && $read_name =~ /\/[12]$/) { + $acc = $read_name; + } + + + if (! exists $scaff_to_partitions{$scaff}) { + # no partitions... should explore why this is. + print STDERR "-warning, no read partitions defined for scaffold: $scaff\n"; + next; + } + + + my $aligned_strand = $sam_entry->get_query_strand(); + my $opposite_strand = ($aligned_strand eq '+') ? '-' : '+'; + + if ($aligned_strand eq '-') { + # restore to actual sequenced bases + $seq = &reverse_complement($seq); + } + + + my $long_read_status = &get_long_read_status($sam_entry); + + + if ($SS_lib_type) { + ## got SS data + + my $transcribed_orient; + + if (! $sam_entry->is_paired()) { + + # if long read, already put in the proper orientation. + + if (! $long_read_status) { + if ($SS_lib_type !~ /^(F|R)$/) { + confess "Error, read is not paired but SS_lib_type set to paired: $SS_lib_type\nread:\n$_"; + } + + if ($SS_lib_type eq "R") { + $seq = &reverse_complement($seq); + } + } + } + + else { + ## Paired reads. + if ($SS_lib_type !~ /^(FR|RF)$/) { + confess "Error, read is paired but SS_lib_type set to unpaired: $SS_lib_type\nread:\n$_"; + } + + my $first_in_pair = $sam_entry->is_first_in_pair(); + if ( ($first_in_pair && $SS_lib_type eq "RF") + || + ( (! $first_in_pair) && $SS_lib_type eq "FR") + ) { + $seq = &reverse_complement($seq); + } + } + } + + + my $new_partition_flag = 0; + + ## prime ordered partitions if first entry or if switching scaffolds. + if ($scaff ne $current_scaff) { + + $current_scaff = $scaff; + + @ordered_partitions = @{$scaff_to_partitions{$scaff}}; + $current_partition = shift @ordered_partitions; + $partition_counter = 0; + + $new_partition_flag = 1; + + + } + elsif ($current_partition && $start > $current_partition->{rend}) { + $current_partition = shift @ordered_partitions; + $partition_counter++; + + $new_partition_flag = 1; + + } + + if ($new_partition_flag) { + + close $ofh if $ofh; + $ofh = undef; + + close $sam_ofh if $sam_ofh; + $sam_ofh = undef; + + if ($read_counter < $MIN_READS_PER_PARTITION) { + # delete these read files. + #print STDERR "-- too few reads ($read_counter), removing partition: $part_file\n"; + unlink($part_file, $sam_part_file); + } + + $read_counter = 0; + + } + + + ## check to see if we're in a partition + if (defined($current_partition) && $start >= $current_partition->{lend} && $start <= $current_partition->{rend}) { + # may need to start a new ofh for this partition if not already established. + unless ($ofh) { + # create new one. + my $file_part_count = int($partition_counter/$PARTS_PER_DIR); + my $outdir = "$partitions_dir/" . $current_partition->{scaff} . "/$file_part_count"; + $outdir =~ s/[\;\|]/_/g; + + mkpath($outdir) if (! -d $outdir); + unless (-d $outdir) { + die "Error, cannot mkdpath $outdir"; + } + + $part_file = "$outdir/" . join("_", $current_partition->{lend}, $current_partition->{rend}) . ".trinity.reads"; + open ($ofh, ">$part_file") or die "Error, cannot write ot $part_file"; + print STDERR "-writing to $part_file\n"; + + print $track_fh join("\t", $scaff, $current_partition->{lend}, $current_partition->{rend}, $part_file) . "\n"; + + $sam_part_file = "$outdir/" . join("_", $current_partition->{lend}, $current_partition->{rend}) . ".sam"; + open ($sam_ofh, ">$sam_part_file") or die "Error, cannot open $sam_part_file"; + + + } + # write to partition + if ($long_read_status) { + $acc = "LR\$|$acc"; + } + print $ofh ">$acc\n$seq\n"; + print $sam_ofh join("\t", $sam_entry->get_fields()) . "\n";; + $read_counter++; + + } + + } + close $track_fh; + + close $ofh if $ofh; + close $sam_ofh if $sam_ofh; + + exit(0); +} + + + +#### +sub parse_partitions { + my ($partitions_file) = @_; + + my %scaff_to_parts; + + print STDERR "// parsing paritions.\n"; + my $counter = 0; + + open (my $fh, $partitions_file) or die "Error, cannot open file $partitions_file"; + while (<$fh>) { + chomp; + if (/^\#/) { next; } + unless (/\w/) { next; } + + $counter++; + print STDERR "\r[$counter] " if $counter % 100 == 0; + + my @x = split(/\t/); + + my $scaff = $x[0]; + my $lend = $x[3]; + my $rend = $x[4]; + my $orient = $x[6]; + + push (@{$scaff_to_parts{$scaff}}, { scaff => $scaff, + lend => $lend, + rend => $rend, } ); + + } + print STDERR "\r[$counter] "; + + close $fh; + + # should be sorted, but let's just be sure: + foreach my $scaff (keys %scaff_to_parts) { + @{$scaff_to_parts{$scaff}} = sort {$a->{lend}<=>$b->{lend}} @{$scaff_to_parts{$scaff}}; + } + + return(%scaff_to_parts); +} + + + +sub get_long_read_status { + my ($sam_entry) = @_; + + + my @fields = $sam_entry->get_fields(); + @fields = @fields[11..$#fields]; + + if (grep { /^RG:Z:PBLR$/ } @fields) { + return(1); + } + else { + return(0); + } +} diff --git a/99.scripts/trinity_utils/util/support_scripts/fastQ_to_fastA.pl b/99.scripts/trinity_utils/util/support_scripts/fastQ_to_fastA.pl new file mode 100644 index 0000000..13484dd --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/fastQ_to_fastA.pl @@ -0,0 +1,120 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Nuc_translator; +use IO::Uncompress::Gunzip; + +use Getopt::Long qw(:config no_ignore_case bundling); + + +my $usage = <<_EOUSAGE_; + +########################################################## +# +# -I input.fq or "input1.fq,input2.fq,input3.fq" +# +# --ignoreDirty ignores poorly formed entries +# +# -a append "/num" to the accession name. +# +# --rev reverse complement nucleotide sequence. +# +########################################################### + +_EOUSAGE_ + ; + +my $inputFile; +my $ignore_dirty = 0; +my $append_num; +my $revcomp_flag = 0; + +&GetOptions( 'I=s' => \$inputFile, + 'ignore_dirty' => \$ignore_dirty, + 'a=i' => \$append_num, + 'rev' => \$revcomp_flag, + ); + + +unless ($inputFile) { + die $usage; +} + +main: { + my @files = split(/,/, $inputFile); + foreach my $file (@files) { + $file =~ s/\s//g; + if ($file =~ /\w/) { + &fastQ_to_fastA($file); + } + } + exit(0); +} + + +sub fastQ_to_fastA { + my ($file) = @_; + + my $fh = new IO::Uncompress::Gunzip($file) or die "Error, cannot open file $file"; + + my $counter = 0; + my $num_clean = 0; + my $num_dirty = 0; + + while (my $line = <$fh>) { + $line =~ s/\cM//g; # remove any cntrl-M characters (sometimes derived from MS-windows text files) + + if ($line =~ /^\@/) { + $counter++; + + # print STDERR "\r[$counter] [$num_clean clean] [$num_dirty dirty] " if ($counter % 10000 == 0); + + my $header = $line; + my $seq = <$fh>; + my $qual_header = <$fh>; + my $qual_line = <$fh>; + + chomp $header; + chomp $seq if $seq; + chomp $qual_header if $qual_header; + chomp $qual_line if $qual_line; + + if ($header && $seq && $qual_header && $qual_line =~ /\S/ && + $qual_header =~ /^\+/ && length($seq) == length($qual_line)) { + + # can do some more checks here if needed to be sure that the lines are formatted as expected. + + substr($header,0,1,''); # strip beginning "@" + + my @header_parts = split(/\s+/, $header); + $header = shift @header_parts; + + if (@header_parts && $header !~ m|/[12]$| && $header_parts[0] =~ /^([12])\:/) { + $header .= "/$1"; + } + if (defined $append_num) { + $header .= "/$append_num"; + } + if ($revcomp_flag) { + $seq = &reverse_complement($seq); + } + + print ">$header\n$seq\n"; + $num_clean++; + } + else { + $num_dirty++; + unless ($ignore_dirty) { + die "Error, improperly formatted entry:\n\n$header\n$seq\n$qual_header\n$qual_line\n"; + } + + } + } + } + + return; +} diff --git a/99.scripts/trinity_utils/util/support_scripts/fastQ_to_tab.pl b/99.scripts/trinity_utils/util/support_scripts/fastQ_to_tab.pl new file mode 100644 index 0000000..6376cdd --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/fastQ_to_tab.pl @@ -0,0 +1,137 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; + +use Getopt::Long qw(:config no_ignore_case bundling); + + +my $usage = <<_EOUSAGE_; + +########################################################## +# +# -I input.fq +# +# --ignore_dirty ignores poorly formed entries +# +# -a append "/num" to the accession name. +# +# -v verbose +# +########################################################### + +_EOUSAGE_ + + ; + +my $inputFile; +my $ignore_dirty = 0; +my $append_num; +my $VERBOSE = 0; + +&GetOptions( 'I=s' => \$inputFile, + 'ignore_dirty' => \$ignore_dirty, + 'a=i' => \$append_num, + 'v' => \$VERBOSE, + ); + + +unless ($inputFile) { + die $usage; +} + + +my $fh; +if ($inputFile =~ /\.gz$/) { + open ($fh, "gunzip -c $inputFile | ") or die $!; +} +else { + open ($fh, $inputFile) or die "Error, cannot open $inputFile"; +} + +my $counter = 0; +my $num_clean = 0; +my $num_dirty = 0; + +my @rec; + +my $line = <$fh>; + +while ($line) { + + if ($line =~ /^\@/) { + $counter++; + + print STDERR "\r[$counter] [$num_clean clean] [$num_dirty dirty] " if ($counter % 10000 == 0 && $VERBOSE); + + push (@rec, $line); + + $line = <$fh>; + for (1..3) { + push (@rec, $line); + $line = <$fh>; + } + + my $record_text = join("", @rec); + + my $header = shift @rec; + my $seq = shift @rec; + my $qual_header = shift @rec; + my $qual_line = shift @rec; + + chomp $header; + chomp $seq if $seq; + chomp $qual_header if $qual_header; + chomp $qual_line if $qual_line; + + + my @header_pts = split(/\s+/, $header); + if (scalar @header_pts > 1) { + $header = shift @header_pts; + } + + if ($header && $seq && $qual_header && $qual_line && + $qual_header =~ /^\+/ && length($seq) == length($qual_line)) { + + # can do some more checks here if needed to be sure that the lines are formatted as expected. + + $header =~ s/^\@//; + + ## convert casava format over + my @pts = split(/\s+/, $header); + if (scalar @pts > 1 && $pts[1] =~ /^([12]):/) { + my $val = $1; + $header = $pts[0] . "/$val"; + } + elsif ($append_num) { + $header .= "/$append_num"; + } + + print "$header\t$seq\t$qual_line\n"; + $num_clean++; + } + else { + + $num_dirty++; + + unless ($ignore_dirty) { + die "Error, improperly formatted entry:\n\n$record_text "; + } + + } + @rec = (); + } else { + $line = <$fh>; + } + +} + +exit(0); + + + + + + diff --git a/99.scripts/trinity_utils/util/support_scripts/fasta_find_duplicates.pl b/99.scripts/trinity_utils/util/support_scripts/fasta_find_duplicates.pl new file mode 100644 index 0000000..d6c6bba --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/fasta_find_duplicates.pl @@ -0,0 +1,46 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use Fasta_reader; + + +my $usage = "usage: $0 seqs.fasta\n\n"; + +my $file = $ARGV[0] or die $usage; + +my %seq_to_header; + +my $fasta_reader = new Fasta_reader($file); +while (my $seq_obj = $fasta_reader->next()) { + + my $sequence = $seq_obj->get_sequence(); + my $header = $seq_obj->get_header(); + + if (exists $seq_to_header{$sequence}) { + push (@{$seq_to_header{$sequence}}, $header); + } + else { + $seq_to_header{$sequence} = [$header]; + } +} + +my $found_dups_flag = 0; +foreach my $sequence (keys %seq_to_header) { + my @dups = @{$seq_to_header{$sequence}}; + + if (scalar(@dups) > 1) { + print "# Repeated seqs found:\n"; + print join("\n", @dups) . "\n"; + print "$sequence\n\n"; + $found_dups_flag = 1; + } + +} + + +exit($found_dups_flag); + + diff --git a/99.scripts/trinity_utils/util/support_scripts/fasta_to_tab.pl b/99.scripts/trinity_utils/util/support_scripts/fasta_to_tab.pl new file mode 100644 index 0000000..b04da09 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/fasta_to_tab.pl @@ -0,0 +1,37 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + +my $usage = "usage: $0 [multiFastaFile] [NO_FULL_HEADER_FLAG=0]\n\n"; + +my $input = $ARGV[0] || *STDIN{IO}; +my $NO_FULL_HEADER_FLAG = $ARGV[1] || 0; + +unless (-f $input || ref $input eq 'IO::Handle') { + die "Error, input not established."; +} + +main: { + + my $fasta_reader = new Fasta_reader($input); + + while (my $seq_obj = $fasta_reader->next()) { + my $sequence = $seq_obj->get_sequence(); + my $header = $seq_obj->get_header(); + + if ($NO_FULL_HEADER_FLAG) { + $header = $seq_obj->get_accession(); + } + + print "$header\t$sequence\n"; + } + + exit(0); +} + + + diff --git a/99.scripts/trinity_utils/util/support_scripts/filter_iworm_by_min_length_or_cov.pl b/99.scripts/trinity_utils/util/support_scripts/filter_iworm_by_min_length_or_cov.pl new file mode 100644 index 0000000..bf99194 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/filter_iworm_by_min_length_or_cov.pl @@ -0,0 +1,35 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + + +my $usage = "usage: $0 IwormFastaFile min_length min_cov\n\n"; + +my $fasta_file = $ARGV[0] or die $usage; +my $min_length = $ARGV[1] or die $usage; +my $min_cov = $ARGV[2] or die $usage; + +my $fasta_reader = new Fasta_reader($fasta_file); + +while (my $seqobj = $fasta_reader->next()) { + my $fasta_entry = $seqobj->get_FASTA_format(); + my $sequence = $seqobj->get_sequence(); + + my $iworm_acc = $seqobj->get_accession(); + my ($iworm_num, $iworm_cov, @rest) = split(/;/, $iworm_acc); + + unless (length($sequence) >= $min_length || $iworm_cov >= $min_cov) { + next; + } + + print $fasta_entry; +} + + +exit(0); + diff --git a/99.scripts/trinity_utils/util/support_scripts/filter_transcripts_require_min_cov.pl b/99.scripts/trinity_utils/util/support_scripts/filter_transcripts_require_min_cov.pl new file mode 100644 index 0000000..5dd1f0f --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/filter_transcripts_require_min_cov.pl @@ -0,0 +1,108 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use Fasta_reader; +use DelimParser; +use Carp; +use Data::Dumper; + +my $usage = "\n\n\tusage: $0 Trinity.fasta reads.fa salmon.quant.sf min_cov\n\n"; + +my $trin_fa = $ARGV[0] or die $usage; +my $reads_fa = $ARGV[1] or die $usage; +my $salmon_quant_file = $ARGV[2] or die $usage; +my $min_cov = $ARGV[3] or die $usage; + +main: { + + my $read_length = &estimate_read_length($reads_fa); + + my %want; + { + open(my $fh, "$salmon_quant_file") or die "Error, cannot open file: $salmon_quant_file"; + my $tab_reader = new DelimParser::Reader($fh, "\t"); + my %col_headers = map { + $_ => 1 } $tab_reader->get_column_headers(); + # Name Length EffectiveLength TPM NumReads + unless (exists $col_headers{'Name'} && exists $col_headers{'NumReads'} && exists $col_headers{'Length'}) { + confess "ERROR, not recognizing salmon header. Need Name, NumReads, and Length columns"; + } + while(my $row = $tab_reader->get_row()) { + my $transcript = $row->{Name}; + my $length = $row->{Length}; + my $count = $row->{NumReads}; + + + unless (defined $transcript) { + confess "Error, no transcript name provided at " . Dumper($row); + } + unless (defined $count) { + confess "Error, no NumReads value provided at " . Dumper($row); + } + unless (defined $length) { + confess "Error, no length value provided at " . Dumper($row); + } + + my $eff_cov = $count * $read_length / $length; + + if ($eff_cov >= $min_cov) { + $want{$transcript} = 1; + } + } + } + + my $fasta_reader = new Fasta_reader($trin_fa); + while (my $seq_obj = $fasta_reader->next()) { + my $acc = $seq_obj->get_accession(); + if ($want{$acc}) { + my $fasta_entry = $seq_obj->get_FASTA_format(); + + print $fasta_entry; + + delete $want{$acc}; + } + + + } + + if (%want) { + confess "Error, missed retrieving entries from fasta file: " . Dumper(%want); + } + + exit(0); +} + + +#### +sub estimate_read_length { + my ($fa_file) = @_; + + my $fasta_reader = new Fasta_reader($fa_file); + + my $max_records = 100; + + my $record_counter = 0; + my $sum_lens = 0; + while (my $seq_obj = $fasta_reader->next()) { + my $sequence = $seq_obj->get_sequence(); + $sum_lens += length($sequence); + + $record_counter++; + if ($record_counter >= $max_records) { + last; + } + } + + my $avg_len = $sum_lens / $record_counter; + + return($avg_len); +} + + + + + + diff --git a/99.scripts/trinity_utils/util/support_scripts/fragment_coverage_writer.pl b/99.scripts/trinity_utils/util/support_scripts/fragment_coverage_writer.pl new file mode 100644 index 0000000..4a94739 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/fragment_coverage_writer.pl @@ -0,0 +1,114 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use SAM_reader; +use SAM_entry; + +my $usage = "usage: $0 file.sam.frag_coords\n\n"; + +my $frag_coords_file = $ARGV[0] or die $usage; + +main: { + + my %scaffold_to_coverage; # will retain all coverage information. + + + open (my $fh, $frag_coords_file) or die "Error, cannot open file $frag_coords_file"; + + my $current_scaff = undef; + + my $counter = 0; + while (my $line = <$fh>) { + + chomp $line; + + $counter++; + + my ($scaff, $frag_name, $lend, $rend) = split(/\t/, $line); + + if ($counter % 1000 == 0) { + print STDERR "\r[$counter lines read] scaff:$scaff lend:$lend rend:$rend "; + } + + if (%scaffold_to_coverage && $scaff ne $current_scaff) { + &report_coverage(\%scaffold_to_coverage); + %scaffold_to_coverage = (); + } + + $current_scaff = $scaff; + + &add_coverage($current_scaff, [$lend, $rend], \%scaffold_to_coverage); + + + } + + if (%scaffold_to_coverage) { + &report_coverage(\%scaffold_to_coverage); + } + + + exit(0); + +} + + +#### +sub report_coverage { + my ($scaffold_to_coverage_href) = @_; + + ## output the coverage information: + foreach my $scaffold (sort keys %$scaffold_to_coverage_href) { + + unless (defined($scaffold) && $scaffold =~ /\w/) { next; } + + print "variableStep chrom=$scaffold\n"; + + my $coverage_aref = $scaffold_to_coverage_href->{$scaffold}; + + my $first_cov_pos = &find_first_cov_pos($coverage_aref); + + for (my $i = $first_cov_pos; $i <= $#$coverage_aref; $i++) { + my $cov = $coverage_aref->[$i] || 0; + + print "$i\t$cov\n"; + } + + } + + return; +} + + +#### +sub add_coverage { + my ($scaffold, $coords_aref, $scaffold_to_coverage_href) = @_; + + my ($lend, $rend) = @$coords_aref; + + ## add coverage: + for (my $i = $lend; $i <= $rend; $i++) { + $scaffold_to_coverage_href->{$scaffold}->[$i]++; + } + + + return; +} + +#### +sub find_first_cov_pos { + my ($coverage_aref) = @_; + + for (my $i = 0; $i <= $#$coverage_aref; $i++) { + if ($coverage_aref->[$i]) { + return($i); + } + } + + return($#$coverage_aref); +} + + diff --git a/99.scripts/trinity_utils/util/support_scripts/get_Trinity_gene_to_trans_map.pl b/99.scripts/trinity_utils/util/support_scripts/get_Trinity_gene_to_trans_map.pl new file mode 100644 index 0000000..4559164 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/get_Trinity_gene_to_trans_map.pl @@ -0,0 +1,28 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + + +while (<>) { + if (/>(\S+)/) { + my $acc = $1; + if ($acc =~ /^(.*c\d+_g\d+)(_i\d+)/) { + my $gene = $1; + my $trans = $1 . $2; + + print "$gene\t$trans\n"; + } + elsif ($acc =~ /^(.*comp\d+_c\d+)/) { + my $gene = $1; + my $trans = $acc; + print "$gene\t$trans\n"; + } + else { + print STDERR "WARNING: cannot decipher accession $acc\n"; + } + } +} + +exit(0); + diff --git a/99.scripts/trinity_utils/util/support_scripts/inchworm_transcript_splitter.pl b/99.scripts/trinity_utils/util/support_scripts/inchworm_transcript_splitter.pl new file mode 100644 index 0000000..6d848ea --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/inchworm_transcript_splitter.pl @@ -0,0 +1,173 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Getopt::Long qw(:config no_ignore_case bundling); + +use FindBin; + +use Cwd; + +$ENV{LC_ALL} = 'C'; + +my $util_dir = "$FindBin::RealBin"; + + +my $usage = <<_EOUSAGE_; + + +######################################################################################### +# +# Required: +# +# --iworm inchworm assembled contigs +# +# --left left fragment file +# --right right fragment file +# +# or --single_but_really_paired single read file containing both pairs. +# +# --seqType fq|fa +# +# Optional (if strand-specific RNA-Seq): +# +# --work_dir directory to perform data processing (default: workdir.\$pid +# +# --SS_lib_type RF or FR +# +# --CPU default: 2 +# +########################################################################################### + +_EOUSAGE_ + +; + + +my $inchworm_contigs; +my $left_file; +my $right_file; +my $single_file; +my $seqType; +my $SS_lib_type; +my $work_dir; +my $CPU = 2; + +&GetOptions( 'iworm=s' => \$inchworm_contigs, + 'left=s' => \$left_file, + 'right=s' => \$right_file, + 'seqType=s' => \$seqType, + 'SS_lib_type=s' => \$SS_lib_type, + 'work_dir=s' => \$work_dir, + 'CPU=i' => \$CPU, + 'single_but_really_paired=s' => \$single_file, + ); + + +unless ($inchworm_contigs && ($single_file || ($left_file && $right_file)) && $seqType) { + die $usage; +} + +unless ($work_dir) { + $work_dir = cwd() . "/jaccard_clip_workdir"; + unless (-d $work_dir) { + mkdir($work_dir) or die "Error, cannot mkdir $work_dir"; + } +} + + +my $SYMLINK = ($ENV{NO_SYMLINK}) ? "cp" : "ln -sf"; + +main: { + + my $curr_dir = cwd(); + + # create full paths to inputs, if not set. + foreach my $file ($inchworm_contigs, $left_file, $right_file, $single_file) { + + unless ($file) { next; } + + unless ($file =~ /^\//) { + $file = "$curr_dir/$file"; + } + } + + my $outdir = $work_dir; + unless (-d $outdir) { + mkdir ($outdir) or die "Error, cannot mkdir $outdir"; + } + + + + my $target_iworm_fa = "$outdir/iworm.fa"; + &process_cmd("$SYMLINK $inchworm_contigs $target_iworm_fa") unless (-e $target_iworm_fa); + + + ## run the bowtie alignment pipeline + + my $bowtie_out = "$outdir"; + my $cmd = ""; + + if ($left_file && $right_file) { + + $cmd = "$util_dir/bowtie2_wrapper.pl --seqType $seqType --left $left_file --right $right_file --CPU $CPU --target $target_iworm_fa -o $bowtie_out/bowtie2"; + } + else { + + $cmd = "$util_dir/bowtie2_wrapper.pl --seqType $seqType --single $single_file --CPU $CPU --target $target_iworm_fa -o $bowtie_out/bowtie2"; + } + + if ($SS_lib_type) { + $cmd .= " --SS_lib_type $SS_lib_type "; + } + + + chdir $outdir or die "Error, cannot cd to $outdir"; + + + &process_cmd($cmd); + + + my $final_bam_file = ($SS_lib_type) ? "$bowtie_out/bowtie2.coordSorted.bam.+.bam" : "$bowtie_out/bowtie2.coordSorted.bam"; + + my $alignment_file = "bowtie_alignments.for_jaccard.bam"; + &process_cmd("$SYMLINK $final_bam_file $alignment_file"); + + ## run Jaccard computation: + my $jaccard_wig_file = "$alignment_file.J100.wig"; + my $frag_coords_file = "$alignment_file.frag_coords"; + &process_cmd("$util_dir/SAM_ordered_pair_jaccard.pl --sam $alignment_file -W 100 > $jaccard_wig_file") unless (-s $jaccard_wig_file && -s $frag_coords_file); + + # The above creates a .frag_coords file. Use this to compute fragment-level coverage + my $frag_coverage_file = "$frag_coords_file.wig"; + &process_cmd("$util_dir/fragment_coverage_writer.pl $frag_coords_file > $frag_coverage_file") unless (-e $frag_coverage_file); + + + ## define the transcript clip points: + my $clips_file = "$jaccard_wig_file.clips"; + &process_cmd("$util_dir/jaccard_wig_clipper.pl --jaccard_wig $jaccard_wig_file --coverage_wig $frag_coverage_file > $clips_file") unless (-e $clips_file) ; + + ## clip the inchworm transcripts: + &process_cmd("$util_dir/jaccard_fasta_clipper.pl $inchworm_contigs $clips_file > $inchworm_contigs.clipped.fa") unless (-e "$inchworm_contigs.clipped.fa"); + + + exit(0); + +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print "CMD: $cmd\n"; + + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return($ret); +} diff --git a/99.scripts/trinity_utils/util/support_scripts/iworm_LR_to_scaff_pairs.pl b/99.scripts/trinity_utils/util/support_scripts/iworm_LR_to_scaff_pairs.pl new file mode 100644 index 0000000..51179d4 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/iworm_LR_to_scaff_pairs.pl @@ -0,0 +1,327 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use List::Util qw(min); + +use Carp; +use Getopt::Long qw(:config posix_default no_ignore_case bundling pass_through); +use Data::Dumper; + +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use Fasta_reader; + +my $MIN_CONTAINMENT = 96; +my $KMER_LENGTH = 25; + +my $usage = <<__EOUSAGE__; + +############################################################################## +# +# Required: +# +# --blast_outfmt6_wlen LR_blastn.outfmt6.wLen file +# +# --iworm_fasta inchworm.K25.L25.fa +# +# Optional: +# +# --min_containment default: $MIN_CONTAINMENT +# +# --kmer_len default: $KMER_LENGTH +# +# --debug +# +############################################################################### + + +__EOUSAGE__ + + ; + +my $help_flag; + +my $DEBUG; +my $blast_outfile; +my $iworm_fasta; + +&GetOptions ( 'h' => \$help_flag, + 'blast_outfmt6_wlen=s' => \$blast_outfile, + 'iworm_fasta=s' => \$iworm_fasta, + 'min_containment=i' => \$MIN_CONTAINMENT, + 'kmer_len=i' => \$KMER_LENGTH, + 'debug' => \$DEBUG, + ); + +if ($help_flag) { + die $usage; +} + +unless ($blast_outfile && $iworm_fasta) { + die $usage; +} + + +my %ACC_TO_KMERS_CACHE; +my %TRANS_SEQS; + +main: { + + my %LR_to_iworm_hits; + + { + open (my $fh, $blast_outfile) or die "Error, cannot open file $blast_outfile"; + while (<$fh>) { + #print; + chomp; + my @x = split(/\t/); + my ($iworm_acc, $LR_acc, + $iworm_end5, $iworm_end3, + $LR_end5, $LR_end3, + $bitscore, + $iworm_containment, + ) = ($x[0], $x[1], + $x[6], $x[7], + $x[8], $x[9], + $x[11], + $x[13], + ); + + my ($LR_lend, $LR_rend) = sort {$a<=>$b} ($LR_end5, $LR_end3); + + push (@{$LR_to_iworm_hits{$LR_acc}}, { iworm_acc => $iworm_acc, + lend => $LR_lend, + rend => $LR_rend, + iworm_containment => $iworm_containment, + }); + + + + } + close $fh; + } + + + my $fasta_reader = new Fasta_reader($iworm_fasta); + %TRANS_SEQS = $fasta_reader->retrieve_all_seqs_hash(); + + + my %iworm_pairs; + + foreach my $LR (keys %LR_to_iworm_hits) { + + my @iworm_hits = sort {$a->{lend}<=>$b->{lend} + || + $a->{rend} <=> $b->{rend}} @{$LR_to_iworm_hits{$LR}}; + + + if ($DEBUG) { + # simple report of iworm hits: + print STDERR "\n// Iworm matches for $LR\n"; + foreach my $iworm_hit (@iworm_hits) { + print STDERR join("\t", $iworm_hit->{iworm_acc}, $iworm_hit->{lend}, + $iworm_hit->{rend}, $iworm_hit->{iworm_containment}) . "\n"; + } + } + + + my %full_containments; + + for (my $i = 0; $i < $#iworm_hits; $i++) { + + my $iworm_i = $iworm_hits[$i]; + + for (my $j = $i + 1; $j <= $#iworm_hits; $j++) { + + my $iworm_j = $iworm_hits[$j]; + + if ($iworm_i->{iworm_acc} eq $iworm_j->{iworm_acc}) { next; } + + if ($iworm_j->{lend} > $iworm_i->{rend}) { + last; + } + + unless ($iworm_i->{iworm_containment} >= $MIN_CONTAINMENT || $iworm_j->{iworm_containment} >= $MIN_CONTAINMENT) { + next; + } + + if ($iworm_i->{iworm_containment} >= $MIN_CONTAINMENT) { + $full_containments{ $iworm_i->{iworm_acc} } = 1; + } + if ($iworm_j->{iworm_containment} >= $MIN_CONTAINMENT) { + $full_containments{ $iworm_j->{iworm_acc} } = 1; + } + + my $overlap_len = &get_overlap_len($iworm_i, $iworm_j); + + my $share_kmer_overlap_flag = &share_kmer($iworm_i->{iworm_acc}, $iworm_j->{iworm_acc}); + + if ($DEBUG) { + print STDERR join("\t", $iworm_i->{iworm_acc}, $iworm_j->{iworm_acc}, + "overlap_len:$overlap_len", "share_kmer_overlap:$share_kmer_overlap_flag") . "\n"; + } + + if ($overlap_len && $share_kmer_overlap_flag) { + + my $pair_token = join("$;", sort ($iworm_i->{iworm_acc}, $iworm_j->{iworm_acc})); + $iworm_pairs{$pair_token} = &get_min_iworm_cov($iworm_i->{iworm_acc}, $iworm_j->{iworm_acc}); + + } + } + } # end of i-vs-j comparisons. + + my @fully_contained_iworms = keys %full_containments; + if (scalar(@fully_contained_iworms) > 1) { + # make all pairwise links + for (my $i = 0; $i < $#fully_contained_iworms; $i++) { + my $iworm_i_acc = $fully_contained_iworms[$i]; + for (my $j = $i + 1; $j <= $#fully_contained_iworms; $j++) { + my $iworm_j_acc = $fully_contained_iworms[$j]; + my $pair_token = join("$;", sort ($iworm_i_acc, $iworm_j_acc)); + $iworm_pairs{$pair_token} = &get_min_iworm_cov($iworm_i_acc, $iworm_j_acc); + } + } + } + } + + + ## get fasta index values for the iworm contigs: + my %iworm_to_fasta_index; + { + my $index = 0; + open (my $fh, $iworm_fasta) or die "Error, cannot open file $iworm_fasta"; + while (<$fh>) { + chomp; + if (/^>(\S+)/) { + my $acc = $1; + $iworm_to_fasta_index{$acc} = $index; + $index++; + } + } + close $fh; + } + + + foreach my $pair (keys %iworm_pairs) { + + my ($iworm_acc_A, $iworm_acc_B) = split(/$;/, $pair); + + # a179572;19 + + # assign glue as the lower of the coverage values: + + my ($coreA, $covA) = split(/;/, $iworm_acc_A); + my ($coreB, $covB) = split(/;/, $iworm_acc_B); + + my $glue = min($covA, $covB); + + # output record + my $iworm_index_A = $iworm_to_fasta_index{$iworm_acc_A}; + + my $iworm_index_B = $iworm_to_fasta_index{$iworm_acc_B}; + + + print join("\t", + $iworm_acc_A, $iworm_index_A, + $iworm_acc_B, $iworm_index_B, + $glue) . "\n"; + } + + + + exit(0); + +} + + +#### +sub get_min_iworm_cov { + my ($acc_i, $acc_j) = @_; + + my ($pref_i, $cov_i) = split(/;/, $acc_i); + + my ($pref_j, $cov_j) = split(/;/, $acc_j); + + if ($cov_i < $cov_j) { + return($cov_i); + } + else { + return($cov_j); + } +} + + + +#### +sub get_overlap_len { + my ($iworm_i, $iworm_j) = @_; + + ## require: + ## + ## ----------------- iworm_i + ## ------------------- iworm_j + ## + ## ---------------------------------- LR + ## + ## overlap by at least K-1 + ## staggered overlaps along LR + ## iworm_i and iworm_j alignment lengths >= 2 * (K-1) + + + my $iworm_i_align_len = $iworm_i->{rend} - $iworm_i->{lend} + 1; + my $iworm_j_align_len = $iworm_j->{rend} - $iworm_j->{lend} + 1; + + + if ($iworm_i->{rend} > $iworm_j->{rend}) { + return(0); + } + + + my $overlap_len = $iworm_i->{rend} - $iworm_j->{lend} + 1; + + return($overlap_len); +} + + +#### +sub share_kmer { + my ($accA, $accB) = @_; + + my $kmers_href_A = &get_kmers($accA); + + my $kmers_href_B = &get_kmers($accB); + + foreach my $kmer (keys %$kmers_href_A) { + if (exists $kmers_href_B->{$kmer}) { + return(1); # YES + } + } + + + return(0); # NO + +} + +#### +sub get_kmers { + my ($acc) = @_; + + if (my $href = $ACC_TO_KMERS_CACHE{$acc}) { + return($href); + } + else { + my $sequence = uc $TRANS_SEQS{$acc}; + + my %kmers; + for (my $i = 0; $i <= length($sequence) - ($KMER_LENGTH-1); $i++) { + + my $kmer = substr($sequence, $i, $KMER_LENGTH-1); + $kmers{$kmer} = 1; + } + $ACC_TO_KMERS_CACHE{$acc} = \%kmers; + + return($ACC_TO_KMERS_CACHE{$acc}); + } + +} diff --git a/99.scripts/trinity_utils/util/support_scripts/jaccard_fasta_clipper.pl b/99.scripts/trinity_utils/util/support_scripts/jaccard_fasta_clipper.pl new file mode 100644 index 0000000..9df66b4 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/jaccard_fasta_clipper.pl @@ -0,0 +1,90 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Fasta_reader; + +my $usage = "usage: $0 transcripts.fasta jaccard_clips.wig\n\n"; + +my $transcript_fa = $ARGV[0] or die $usage; +my $jaccard_clips = $ARGV[1] or die $usage; + + + +main: { + + my %trans_acc_to_clips = &get_clip_pts($jaccard_clips); + + my $fasta_reader = new Fasta_reader($transcript_fa); + + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + my $header = $seq_obj->get_header(); + my $seq = $seq_obj->get_sequence(); + + my @parts = split(/;/, $acc); + my $kmer_cov = pop @parts; + + my ($header_begin, $rest_header) = split(/\s+/, $header, 2); + + if (my $clips_aref = $trans_acc_to_clips{$acc}) { + + ## + my $start = 1; + while (@$clips_aref) { + my $clip = shift @$clips_aref; + my $length = $clip - $start + 1; + my $subseq = substr($seq, $start-1, $length); + print ">$acc.$start-$clip;$kmer_cov $rest_header\n$subseq\n"; + $start = $clip + 1; + } + if ($start < length($seq) + 25) { + # require subseq to be at least 25 bases long (one kmer length) + my $subseq = substr($seq, $start-1, length($seq)-$start + 1); + print ">$acc.$start-" . length($seq) . ";$kmer_cov $rest_header\n$subseq\n"; # coverage value needs to be the last piece of the accession for use by PASA + } + } + else { + # no clippint + print ">$header\n$seq\n"; + } + } + + + exit(0); + +} + + +#### +sub get_clip_pts { + my ($jaccard_clips) = @_; + + my %trans_to_clips; + + my $trans_acc = ""; + + open (my $fh, $jaccard_clips) or die "Error, cannot open file $jaccard_clips"; + while (<$fh>) { + chomp; + if (/^variableStep chrom=(\S+)/) { + $trans_acc = $1; + } + elsif (/^(\d+)/) { + my ($coord, $val) = split(/\t/); + if ($val) { + push (@{$trans_to_clips{$trans_acc}}, $coord); + } + } + } + close $fh; + + return(%trans_to_clips); +} + + + diff --git a/99.scripts/trinity_utils/util/support_scripts/jaccard_wig_clipper.pl b/99.scripts/trinity_utils/util/support_scripts/jaccard_wig_clipper.pl new file mode 100644 index 0000000..7d8b428 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/jaccard_wig_clipper.pl @@ -0,0 +1,288 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use WigParser; + +my $jaccard_file; +my $trough_win = 200; +my $min_jaccard_delta = 0.35; +my $max_jaccard_trough_val = 0.05; +my $coverage_wig; + +my $usage = <<__EOUSAGE__; + +############################################################################################################################## +# +# Required; +# +# --jaccard_wig :jaccard wig file +# +# Optional: +# +# --trough_win :default($trough_win) +# --min_jaccard_delta :default($min_jaccard_delta) +# --max_jaccard_trough_val :default($max_jaccard_trough_val) +# +# --coverage_wig :fragment coverage wig. Smallest coverage value in trough_win of putative clip is used. +# +# +############################################################################################################################## + + +__EOUSAGE__ + + ; + + + + +my $VERBOSE = 0; + +my $help_flag; + +&GetOptions ( 'h' => \$help_flag, + 'jaccard_wig=s' => \$jaccard_file, + + 'trough_win=i' => \$trough_win, + 'min_jaccard_delta=f' => \$min_jaccard_delta, + 'max_jaccard_trough_val=f' => \$max_jaccard_trough_val, + + 'coverage_wig=s' => \$coverage_wig, + + ); + + +unless ($jaccard_file) { + die $usage; +} + +if ($help_flag) { + die $usage; +} + +if (@ARGV) { + die "Error, couldn't parse params: @ARGV"; +} + + +main: { + + + my $scaff_to_jaccard_parser = new WigParser($jaccard_file); + + my $scaff_to_frag_coverage_parser; + + if ($coverage_wig) { + $scaff_to_frag_coverage_parser = new WigParser($coverage_wig); + } + + foreach my $scaff (sort $scaff_to_jaccard_parser->get_contig_list()) { + + my @jaccard_vals = $scaff_to_jaccard_parser->get_wig_array($scaff); + + my $vals_aref = \@jaccard_vals; # retaining old usage after module refactoring + + + my @trough_positions; + + for (my $i = $trough_win; $i <= $#$vals_aref; $i++) { + + my $left_win_pos = $i - $trough_win; + my $right_win_pos = $i; + my $center_win = int( ($left_win_pos + $right_win_pos)/2); + + my $jaccard_val_left = $vals_aref->[$left_win_pos]; + my $jaccard_val_right = $vals_aref->[$right_win_pos]; + my $jaccard_val_center = $vals_aref->[$center_win]; + + + if ($jaccard_val_center <= $max_jaccard_trough_val) { + + ## found a trough + push (@trough_positions, { pos => $center_win, + jaccard => $jaccard_val_center, + avg_delta => ( ($jaccard_val_left - $jaccard_val_center ) + ($jaccard_val_right - $jaccard_val_center) ) / 2, + } ); + } + + } + + if (@trough_positions) { + @trough_positions = &group_trough_positions_within_window_select_best_clip(@trough_positions); + + ## restrict clips to those that are in troughs of required depth from neighboring hills + my @clips = &require_neighboring_hills(\@trough_positions, $vals_aref); + + if ($coverage_wig) { + + my @coverage_array = $scaff_to_frag_coverage_parser->get_wig_array($scaff); + + &reposition_clips_by_min_coverage(\@clips, \@coverage_array); + } + + + print "variableStep chrom=$scaff\n"; + foreach my $clip (@clips) { + my $pos = $clip->{pos}; + + print join("\t", $pos, 1) . "\n"; + + } + } + + } + + + exit(0); + + +} + + +#### +sub group_trough_positions_within_window_select_best_clip { + my @trough_positions = @_; + + my @trough_groups; + + my $trough_pos = shift @trough_positions; + push (@trough_groups, [$trough_pos]); + + while (@trough_positions) { + + my $curr_trough_entry = shift @trough_positions; + my $curr_trough_pos = $curr_trough_entry->{pos}; + + + my $prev_trough_group_aref = $trough_groups[$#trough_groups]; + my $last_pos_val = $prev_trough_group_aref->[$#$prev_trough_group_aref]->{pos}; + + if ($curr_trough_pos - $last_pos_val <= $trough_win) { + + push (@$prev_trough_group_aref, $curr_trough_entry); + } + else { + ## start a new group + push (@trough_groups, [ $curr_trough_entry ] ); + } + } + + + ## take the best one as a clip point + ## best defined here as the one with the least jaccard support and greatest delta value + + my @clips; + + foreach my $group (@trough_groups) { + + my @eles = @$group; + + @eles = sort {$a->{jaccard}<=>$b->{jaccard} + || + $b->{avg_delta} <=> $a->{avg_delta} ## order should be low jaccard, high avg_delta + } @eles; + + my $clip_ele = shift @eles; + + push (@clips, $clip_ele); + } + + + return(@clips); +} + + +#### +sub require_neighboring_hills { + my ($clips_aref, $jaccard_aref) = @_; + + my @validated_clips; + + + foreach my $clip (@$clips_aref) { + my $pos = $clip->{pos}; + my $jaccard = $clip->{jaccard}; + + my $hill_jaccard_min = $jaccard + $min_jaccard_delta; + + + my $left_hill_stop = $pos - $trough_win; + if ($left_hill_stop < 1) { + $left_hill_stop = 1; + } + + my $right_hill_stop = $pos + $trough_win; + if ($right_hill_stop > $#$jaccard_aref) { + $right_hill_stop = $#$jaccard_aref; + } + + if (&search_for_hill($jaccard_aref, $left_hill_stop, $pos-1, $hill_jaccard_min) + && + &search_for_hill($jaccard_aref, $pos + 1, $right_hill_stop, $hill_jaccard_min) ) { + + push (@validated_clips, $clip); + } + } + + + return(@validated_clips); +} + +#### +sub search_for_hill { + my ($jaccard_aref, $lend, $rend, $min_jaccard_val) = @_; + + for (my $i = $lend; $i <= $rend; $i++) { + if ($jaccard_aref->[$i] >= $min_jaccard_val) { + return(1); + } + } + + return(0); +} + + +#### +sub reposition_clips_by_min_coverage { + my ($clips_aref, $coverage_aref) = @_; + + foreach my $clip (@$clips_aref) { + + my $clip_pos = $clip->{pos}; + + my $win_left = $clip_pos - int($trough_win/2); + if ($win_left < 1) { + $win_left = 1; + } + my $win_right = $clip_pos + int($trough_win/2); + if ($win_right > $#$coverage_aref) { + $win_right = $#$coverage_aref; + } + + + my $min_cov_val = $coverage_aref->[$clip_pos]; + my $min_cov_pos = $clip_pos; + + for (my $i = $win_left; $i <= $win_right; $i++) { + my $cov = $coverage_aref->[$i]; + if ($cov < $min_cov_val) { + $min_cov_val = $cov; + $min_cov_pos = $i; + } + } + $clip->{pos} = $min_cov_pos; + } + + return; +} + + + diff --git a/99.scripts/trinity_utils/util/support_scripts/join_partitions_within_range.pl b/99.scripts/trinity_utils/util/support_scripts/join_partitions_within_range.pl new file mode 100644 index 0000000..fd4a22c --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/join_partitions_within_range.pl @@ -0,0 +1,42 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "usage: $0 partitions.gff join_within_dist\n"; + +my $partitions_gff = $ARGV[0] or die $usage; +my $join_within_dist = $ARGV[1] or die $usage; + +my @prev_line; + +open (my $fh, $partitions_gff) or die "Error, cannot open file $partitions_gff"; +while (<$fh>) { + chomp; + + my @x = split(/\t/); + + if (!@prev_line) { + @prev_line = @x; + next; + } + + + if ($x[0] ne $prev_line[0] + || + $x[3] - $prev_line[4] > $join_within_dist) { + + print join("\t", @prev_line) . "\n"; + @prev_line = @x; + } + else { + # same contig and within distance + $prev_line[4] = $x[4]; # update right end. + } +} + +print join("\t", @prev_line) . "\n" if @prev_line; + + +exit(0); + diff --git a/99.scripts/trinity_utils/util/support_scripts/kallisto_trans_to_gene_results.pl b/99.scripts/trinity_utils/util/support_scripts/kallisto_trans_to_gene_results.pl new file mode 100644 index 0000000..594bc06 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/kallisto_trans_to_gene_results.pl @@ -0,0 +1,155 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Data::Dumper; + +my $usage = "\n\nusage: $0 abundance.tsv gene_to_trans_map_file.txt\n\n\n"; + +my $abundance_tsv = $ARGV[0] or die $usage; +my $gene_to_trans_map_file = $ARGV[1] or die $usage; + + +main: { + + my %trans_to_gene_info; + { + open (my $fh, $gene_to_trans_map_file) or die "Error, cannot open file $gene_to_trans_map_file"; + while (<$fh>) { + unless (/\w/) { next; } + chomp; + my ($gene, $trans, @rest) = split(/\s+/); + unless ($gene && $trans) { + die "Error, cannot extract gene & trans relationship from line $_ of file $gene_to_trans_map_file"; + } + $trans_to_gene_info{$trans} = $gene; + } + close $fh; + } + + + open (my $fh, $abundance_tsv) or die "Error, cannot open file $abundance_tsv"; + my $header = <$fh>; + chomp $header; + my %field_index; + my @fields = split(/\t/, $header); + { + + for (my $i = 0; $i <= $#fields; $i++) { + my $field = $fields[$i]; + $field_index{$field} = $i; + } + } + + + my %gene_data; + while (<$fh>) { + chomp; + my @x = split(/\t/); + + my $trans_id = $x[ $field_index{target_id} ]; + my $tpm = $x[ $field_index{tpm} ]; + my $length = $x[ $field_index{length} ]; + my $eff_length = $x[ $field_index{eff_length} ]; + my $est_counts = $x[ $field_index{est_counts} ]; + + my $gene = $trans_to_gene_info{$trans_id} or die "Error, cannot find gene identifier for transcript [$trans_id] "; + + push (@{$gene_data{$gene}}, { trans_id => $trans_id, + tpm => $tpm, + length => $length, + eff_length => $eff_length, + est_counts => $est_counts, + }); + + + } + close $fh; + + + ## Output gene summaries: + + print $header . "\n"; + + foreach my $gene (keys %gene_data) { + my @trans_structs = @{$gene_data{$gene}}; + + my @trans_ids; + my $sum_counts = 0; + my $sum_tpm = 0; + + my $counts_per_len_sum = 0; + my $counts_per_eff_len_sum = 0; + + my $sum_lengths = 0; + my $sum_eff_lengths = 0; + + my $num_trans = scalar(@trans_structs); + + + foreach my $struct (@trans_structs) { + + #print Dumper($struct); + + my $trans_id = $struct->{trans_id}; + my $tpm = $struct->{tpm}; + my $length = $struct->{length}; + + my $eff_length = $struct->{eff_length}; + my $est_counts = $struct->{est_counts}; + + unless ($eff_length > 0) { + $eff_length = 1; # cannot have zero length feature! + } + + unless ($length > 0 && $eff_length > 0) { + die "Error, length: $length, eff_length: $eff_length" . Dumper($struct); + } + + $sum_lengths += $length; + $sum_eff_lengths += $eff_length; + + $counts_per_len_sum += $est_counts/$length; + + $counts_per_eff_len_sum += $est_counts/$eff_length; + + $sum_counts += $est_counts; + $sum_tpm += $tpm; + } + + my $gene_length = $sum_lengths / $num_trans; + my $gene_eff_length = $sum_eff_lengths / $num_trans; + if ($sum_counts) { + # set lengths as weighted by expression of isoforms. + eval { + $gene_length = $sum_counts / $counts_per_len_sum; + $gene_eff_length = $sum_counts / $counts_per_eff_len_sum; + }; + if ($@) { + print STDERR "$@\n" . Dumper(\@trans_structs); + die; + } + } + + my %gene_info = ( target_id => $gene, + tpm => sprintf("%.2f", $sum_tpm), + length => sprintf("%.2f", $gene_length), + eff_length => sprintf("%.2f", $gene_eff_length), + est_counts => sprintf("%.2f", $sum_counts), + + ); + + my @vals; + foreach my $field (@fields) { + my $result = $gene_info{$field}; + unless (defined $result) { + $result = "NA"; + } + push (@vals, $result); + } + print join("\t", @vals) . "\n"; + } + + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/support_scripts/merge_pair_and_LR_scaff_links.pl b/99.scripts/trinity_utils/util/support_scripts/merge_pair_and_LR_scaff_links.pl new file mode 100644 index 0000000..b97d302 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/merge_pair_and_LR_scaff_links.pl @@ -0,0 +1,49 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +my $usage = "\n\n\tusage: $0 scaff_pairs_A.txt scaff_pairs_B.txt ...\n\n"; + +my @pair_files = @ARGV; +unless (scalar(@pair_files) > 1) { + die $usage; +} + +main: { + + my %top_pairs; + foreach my $pair_file (@pair_files) { + open (my $fh, $pair_file) or die "Error, cannot open file $pair_file"; + while (<$fh>) { + my $line = $_; + chomp; + + my ($iworm_A, $idx_A, $iworm_B, $idx_B, $count) = split(/\t/); + + my $pair_token = join("$;", sort ($iworm_A, $iworm_B)); + + if ( (! exists $top_pairs{$pair_token}) + || + $top_pairs{$pair_token}->{count} < $count) { + + $top_pairs{$pair_token} = { count => $count, + line => $line }; + + } + } + close $fh; + } + + + my @top_structs = reverse sort {$a->{count} <=> $b->{count}} values %top_pairs; + + foreach my $struct (@top_structs) { + + print $struct->{line}; + + } + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/nbkc_merge_left_right_stats.pl b/99.scripts/trinity_utils/util/support_scripts/nbkc_merge_left_right_stats.pl new file mode 100644 index 0000000..f56aa14 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/nbkc_merge_left_right_stats.pl @@ -0,0 +1,152 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use DelimParser; + +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + +my $usage = <<__EOUSAGE__; + +############################################################################### +# +# Required: +# +# --left left.fq.stats +# --right right.fq.stats +# +# Optional +# +# --sorted flag indicating that entries are lexically sorted +# (this can account for differences in representation by +# reads in either file) +# Unpaired entries are ignored. +# +################################################################################ + +__EOUSAGE__ + + ; + +my $left_stats_file; +my $right_stats_file; + +my $sorted_flag = 0; + +&GetOptions( 'left=s' => \$left_stats_file, + 'right=s' => \$right_stats_file, + 'sorted' => \$sorted_flag, + ); + + +unless ($left_stats_file && $right_stats_file) { + die $usage; +} + + +main: { + + my ($left_fh, $right_fh); + print STDERR "-opening $left_stats_file\n"; + if ($left_stats_file =~ /\.gz$/) { + open ($left_fh, "gunzip -c $left_stats_file | ") or die $!; + } elsif ($left_stats_file =~ /\.xz$/) { + open(${left_fh}, "xz -cd ${left_stats_file} | ") or die $!; + } + else { + open ($left_fh, $left_stats_file) or die $!; + } + + print STDERR "-opening $right_stats_file\n"; + if ($right_stats_file =~ /\.gz$/) { + open ($right_fh, "gunzip -c $right_stats_file | ") or die $!; + } elsif ($right_stats_file =~ /\.xz$/) { + open (${right_fh}, "xz -dc ${right_stats_file} | ") or die $!; + } + else { + open ($right_fh, $right_stats_file) or die $!; + } + + print STDERR "-done opening files.\n"; + + my $left_delim_parser = new DelimParser::Reader($left_fh, "\t"); + my $right_delim_parser = new DelimParser::Reader($right_fh, "\t"); + + + # print header: + print join("\t", "acc", + "left_acc", "left_median_cov", "left_mean_cov", "left_stdev", + "right_acc", "right_median_cov", "right_mean_cov", "right_stdev", + "median_cov", "mean_cov", "stdev" + ) . "\n"; + + my $left_row = $left_delim_parser->get_row(); + my $right_row = $right_delim_parser->get_row(); + + while (1) { + + if ( (! $left_row) || (! $right_row)) { + last; + } + + my $left_acc = $left_row->{acc}; + my $left_median = $left_row->{median_cov}; + my $left_mean = $left_row->{mean_cov}; + my $left_stdev = $left_row->{stdev}; + + my $right_acc = $right_row->{acc}; + my $right_median = $right_row->{median_cov}; + my $right_mean = $right_row->{mean_cov}; + my $right_stdev = $right_row->{stdev}; + + + my $core_acc = $left_acc; + if ($left_acc =~ /^(\S+)\/\d$/) { + $core_acc = $1; + } + + my $right_core = $right_acc; + if ($right_acc =~ /^(\S+)\/\d$/) { + $right_core = $1; + } + + unless ($right_core eq $core_acc) { + + if ($sorted_flag) { + if ($left_acc lt $right_acc) { + $left_row = $left_delim_parser->get_row(); # advance left + } + else { + # advance right + $right_row = $right_delim_parser->get_row(); + } + next; + } + else { + die "Error, core accs are not equivalent: [$core_acc] vs. [$right_core] reads, and --sorted flag wasn't used here."; + } + } + + + print join("\t", $core_acc, + $left_acc, $left_median, $left_mean, $left_stdev, + $right_acc, $right_median, $right_mean, $right_stdev, + sprintf("%.1f", ($left_median + $right_median)/2), + sprintf("%.1f", ($left_mean + $right_mean)/2), + sprintf("%.1f", ($left_stdev + $right_stdev)/2) + ) . "\n"; + + + ## advance next entries. + + $left_row = $left_delim_parser->get_row(); + $right_row = $right_delim_parser->get_row(); + + } + + exit(0); +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/nbkc_normalize.pl b/99.scripts/trinity_utils/util/support_scripts/nbkc_normalize.pl new file mode 100644 index 0000000..0f870f5 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/nbkc_normalize.pl @@ -0,0 +1,129 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); + +use DelimParser; + +use Getopt::Long qw(:config posix_default no_ignore_case bundling pass_through); + + +my $pair_stats_file; +my $max_cov; +my $min_cov=1; +my $max_CV; + + +my $usage = <<__EOUSAGE__; + +############################################################# +# +# Required: +# +# --stats_file : pairs.stats.sorted +# +# --max_cov : maximum coverage +# +# --min_cov : minimum coverage +# +# --max_CV : maximum coeff. var. +# +# +############################################################# + +__EOUSAGE__ + + ; + + + +my $help_flag; + +&GetOptions ( 'h' => \$help_flag, + + 'stats_file=s' => \$pair_stats_file, + 'max_cov=i' => \$max_cov, + 'min_cov=i' => \$min_cov, + 'max_CV=i' => \$max_CV, + + +); + + +unless ($pair_stats_file && + $max_cov && + defined($min_cov) && + defined($max_CV)) { + + die $usage; + +} + + + +main: { + + srand(12345); + + my $count_aberrant_and_discarded = 0; + my $count_selected = 0; + my $count_total = 0; + + my $count_below_min_cov = 0; + + open (my $fh, $pair_stats_file) or die $!; + + my $delim_parser = new DelimParser::Reader($fh, "\t"); + + while (my $row = $delim_parser->get_row()) { + + $count_total++; + + my $core_acc = $row->{acc}; + + $core_acc =~ s|/[12]$||; + + my $med_cov = ($row->{median_cov}); + + if ($med_cov < $min_cov) { + $count_below_min_cov++; + next; + } + + my $sd = $row->{stdev}; + my $u = $row->{mean_cov}; + + if ($u <= 0) { + $count_aberrant_and_discarded++; + next; + } + + my $cv = $sd/$u; + + if ($cv > $max_CV) { + $count_aberrant_and_discarded++; + next; + } + + if (rand(1) <= $max_cov/$med_cov) { + print "$core_acc\n"; + $count_selected++; + } + } + close $fh; + + unless ($count_total) { + die "Error, no reads made it to the normalization process... "; + } + + print STDERR "$count_selected / $count_total = " . sprintf("%.2f", $count_selected/$count_total*100) . "% reads selected during normalization.\n"; + print STDERR "$count_aberrant_and_discarded / $count_total = " . sprintf("%.2f", $count_aberrant_and_discarded/$count_total*100) . "% reads discarded as likely aberrant based on coverage profiles.\n"; + print STDERR "$count_below_min_cov / $count_total = " . sprintf("%.2f", $count_below_min_cov/$count_total*100) . "% reads discarded as below minimum coverage threshold=$min_cov\n"; + + exit(0); +} + + diff --git a/99.scripts/trinity_utils/util/support_scripts/ordered_fragment_coords_to_jaccard.pl b/99.scripts/trinity_utils/util/support_scripts/ordered_fragment_coords_to_jaccard.pl new file mode 100644 index 0000000..9b14222 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/ordered_fragment_coords_to_jaccard.pl @@ -0,0 +1,667 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::RealBin/../../PerlLib"); + +use SAM_reader; +use SAM_entry; + +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + +use Carp; +use Data::Dumper; + +$ENV{LC_ALL} = 'C'; # critical for proper sorting using [system "sort -k1,1 ..."] within the perl script + + +my $usage = <<_EOUSAGE_; + + + +####################################################################### +# +# Required: +# +# --lend_sorted_frags lend-coordinate sorted fragment file +# +# Optional: +# +# -W window length (default: 100) +# -M min number of fragments within window (default: 0) +# --pseudocounts default(1) +# --full fully descriptive output format, rather than wig (default) +# --full_extreme includes fragment names and coordinates +# +# -e extended format (include single count and both count in wig format) +# -v verbose, for progress-monitoring +# +# -d debug mode +# +####################################################################### + + + + +_EOUSAGE_ + + ; + + +my $lend_sorted_frags_file; +my $window_length = 100; +my $full_flag = 0; +my $extra = 0; +my $full_extreme_flag = 0; +my $VERBOSE = 0; +my $DEBUG = 0; +my $help_flag = 0; +my $pseudocounts = 1; +my $extended_flag = 0; + + +my $MIN_FRAGS = 0; + +&GetOptions( + 'h' => \$help_flag, + + 'lend_sorted_frags=s' => \$lend_sorted_frags_file, + "W=i" => \$window_length, + + 'full' => \$full_flag, + 'full_extreme' => \$full_extreme_flag, + 'X' => \$extra, + + 'e' => \$extended_flag, + + 'pseudocounts=i' => \$pseudocounts, + + 'v' => \$VERBOSE, + 'd' => \$DEBUG, + 'M=i' => \$MIN_FRAGS, + ); + +if ($help_flag) { + die $usage; +} + + +unless ($lend_sorted_frags_file) { + die $usage; +} + +if ($full_extreme_flag) { + $full_flag = 1; +} + +if (@ARGV) { + die $usage; +} + +main: { + + + ## sort by rend + my $rend_sorted_frags_file = "$lend_sorted_frags_file.sort_by_rend"; + my $cmd = "sort -T . -k1,1 -k4,4n $lend_sorted_frags_file > $rend_sorted_frags_file"; + &process_cmd($cmd) unless (-s "$rend_sorted_frags_file"); + + print STDERR "-processing jaccard pair sensor\n"; + &compute_jaccard_wig($lend_sorted_frags_file, $rend_sorted_frags_file, $window_length); + + + exit(0); + + + +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret $ret"; + } + + return; +} + + + +#### +sub compute_jaccard_wig { + my ($frags_sort_lend_file, $frags_sort_rend_file, $window_length) = @_; + + ## maintain two points, window length apart. + ## at each point: track accessions entering and leaving at each data point. + + + ## Left window scanner + my $left_window_scan_LEFT = Read_reader->new($frags_sort_lend_file); + my $left_window_scan_RIGHT = Read_reader->new($frags_sort_rend_file); + + my $right_window_scan_LEFT = Read_reader->new($frags_sort_lend_file); + my $right_window_scan_RIGHT = Read_reader->new($frags_sort_rend_file); + + my $curr_molecule = $left_window_scan_LEFT->get_curr_scaffold(); + print "variableStep chrom=$curr_molecule\n"; + + my $window_lend = $left_window_scan_LEFT->get_curr_read_lend(); + my $window_rend = $window_lend + $window_length; + + my %frag_counter; + + my $num_single = 0; + my $num_both = 0; + + my %extreme_verbose_frag_capture; + my $prev_pos = 0; + my $prev_OK = 0; + + my @rend_tracker; + + while ($left_window_scan_LEFT->get_curr_line()) { + + print STDERR "\r[$window_lend] " if $VERBOSE; + + + ## see if we need to move on to the next molecule + if ($left_window_scan_LEFT->get_curr_scaffold() gt $curr_molecule + && + $left_window_scan_LEFT->get_curr_scaffold() eq $right_window_scan_LEFT->get_curr_scaffold()) { + + + # init for next molecule + $curr_molecule = $left_window_scan_LEFT->get_curr_scaffold(); + print "variableStep chrom=$curr_molecule\n"; + + ## advance the other readers to ensure they're all synched up to the same molecule. + $left_window_scan_RIGHT->advance_to_scaffold($curr_molecule); + $right_window_scan_LEFT->advance_to_scaffold($curr_molecule); + $right_window_scan_RIGHT->advance_to_scaffold($curr_molecule); + + + %frag_counter = (); + $num_single = 0; + $num_both = 0; + %extreme_verbose_frag_capture = (); + $prev_pos = 0; + @rend_tracker = (); + + #if ($window_lend < 1) { + # $window_lend = 1; + #} + + $window_lend = $left_window_scan_LEFT->get_curr_read_lend(); + } + + + my $next_cand_lend = $left_window_scan_LEFT->get_curr_read_lend(); + my $rend = $left_window_scan_LEFT->get_curr_read_rend(); + + push (@rend_tracker, $rend); + @rend_tracker = sort {$a<=>$b} @rend_tracker; + while (@rend_tracker && $rend_tracker[0] <= $window_lend) { + shift @rend_tracker; + } + if ($rend_tracker[0] < $next_cand_lend) { + $window_lend = $rend_tracker[0]; + } + else { + $window_lend = $next_cand_lend; + } + + $window_rend = $window_lend + $window_length - 1; + + { + + ## Fragment entering Right window + ## process RIGHT window, lend of frag + + my @acc_infos = $right_window_scan_LEFT->advance_get_accessions($curr_molecule, $window_rend, "LEND"); + foreach my $acc_info (@acc_infos) { + my ($acc, $lend, $rend) = @$acc_info; + my $count = ++$frag_counter{$acc}; + if ($count == 1) { + $num_single++; + } + else { + die "Error, lend of frag $acc entering REND marker, should be seen for first time, but count is: $count"; + } + if ($DEBUG) { + print "WINRIGHT\tLEND\t$acc\t$window_rend\t$count\n"; + } + $extreme_verbose_frag_capture{$acc} = "$acc\[$lend-$rend]" if ($full_extreme_flag); + + + } + } + + { + + ## Fragment exiting right window + ## process right window, rend of frag + + my @acc_infos = $right_window_scan_RIGHT->advance_get_accessions($curr_molecule, $window_rend-1, "REND"); + foreach my $acc_info (@acc_infos) { + my ($acc, $lend, $rend) = @$acc_info; + if (exists $frag_counter{$acc}) { + my $count = --$frag_counter{$acc}; + if ($count == 1) { + $num_single++; + $num_both--; + } + elsif ($count == 0) { + $num_single--; + delete $frag_counter{$acc}; + } + else { + die "Error, count of $acc is $count"; + } + + if ($DEBUG) { + print "WINRIGHT\tREND\t$acc\t$window_rend\t$count\n"; + } + $extreme_verbose_frag_capture{$acc} = "$acc\[$lend-$rend]" if ($full_extreme_flag); + } + else { + die "Error, frag $acc exiting rend marker and hasn't been logged"; + } + } + } + + { + + ## Fragment entering left + ## process left window, lend of frag + + my @acc_infos = $left_window_scan_LEFT->advance_get_accessions($curr_molecule, $window_lend, "LEND"); + foreach my $acc_info (@acc_infos) { + my ($acc, $lend, $rend) = @$acc_info; + + my $count = ++$frag_counter{$acc}; + if ($count == 2) { + # spans both window markers + $num_single--; + $num_both++; + } + elsif ($count == 1) { + $num_single++; + } + else { + die "weird error, count: $count"; + } + if ($DEBUG) { + print "WINLEFT\tLEND\t$acc\t$window_lend\t$count\n"; + } + + $extreme_verbose_frag_capture{$acc} = "$acc\[$lend-$rend]" if ($full_extreme_flag); + + } + } + + { + + ## Fragment exiting left + ## process left window, rend of frag + + my @acc_infos = $left_window_scan_RIGHT->advance_get_accessions($curr_molecule, $window_lend-1, "REND"); + foreach my $acc_info (@acc_infos) { + my ($acc, $lend, $rend) = @$acc_info; + if (exists $frag_counter{$acc}) { + my $count = --$frag_counter{$acc}; + if ($count == 0) { + $num_single--; + delete $frag_counter{$acc}; + } + else { + die "Error, frag: $acc rend passed left edge of window and count is: $count"; + } + if ($DEBUG) { + print "WINLEFT\tREND\t$acc\t$window_lend\t$count\n"; + } + $extreme_verbose_frag_capture{$acc} = "$acc\[$lend-$rend]" if ($full_extreme_flag); + } + else { + die "Error, left marker has frag passing thats not logged."; + } + } + } + + + ## compute jaccard coeff + if ($window_lend >= 1) { # && ($num_single || $num_both)) { + + #my $jaccard = $num_both / ($num_single + $num_both); + my $jaccard = ($num_both + $pseudocounts) / ($num_single + $num_both + $pseudocounts); + + $jaccard = sprintf("%.4f", $jaccard); + my $mid = int( ($window_lend + $window_rend) / 2); + + if ($full_flag && ($num_single + $num_both >= $MIN_FRAGS)) { + # include counts of single and paired frags + print join("\t", $curr_molecule, $mid, $jaccard, "S:$num_single", "P:$num_both"); + if ($full_extreme_flag) { + my @paired; + my @single; + foreach my $frag (keys %frag_counter) { + if ($frag_counter{$frag} == 2) { + push (@paired, $frag); + } + elsif ($frag_counter{$frag} == 1) { + push (@single, $frag); + } + else { + die "Error, count not 1 or 2: " . Dumper(\%frag_counter); + } + } + + if ($num_single != scalar(@single) || $num_both != scalar(@paired)) { + die "Error, inconsistent counts: single: $num_single vs. " . scalar(@single) . ", and paired: $num_both vs. " . scalar(@paired); + } + + &print_frag_info(\@single, \@paired, \%extreme_verbose_frag_capture, $window_lend, $window_rend); + + + } + print "\n"; + } + else { + if ($num_single + $num_both >= $MIN_FRAGS) { + + ## follow up from previous position if in a stretch of supported region + if ($prev_OK && $prev_pos > 0) { + for (my $i = $prev_pos+1; $i < $mid; $i++) { + print join("\t", $i, $jaccard); + if ($extended_flag) { + print "\t" . join("\t", $num_single, $num_both); + } + + if ($extra) { + print "\tS:$num_single P:$num_both **extended"; + } + print "\n"; + } + } + + print join("\t", $mid, $jaccard); + if ($extended_flag) { + print "\t" . join("\t", $num_single, $num_both); + } + if ($extra) { + print "\tS:$num_single P:$num_both"; + } + print "\n"; + $prev_OK = 1; + $prev_pos = $mid; + } + else { + $prev_OK = 0; + } + } + } + + + ## slide window + + #$window_rend++; + #$window_lend = $window_rend - $window_length; + #if ($window_lend < 1) { + # $window_lend = 1; + #} + + + } + + + return; + +} + + +#### +sub print_frag_info { + my ($single_aref, $paired_aref, $frag_info_href, $window_lend, $window_rend) = @_; + + my @left_single; + my @right_single; + + foreach my $frag (@$single_aref) { + my $frag_info = $frag_info_href->{$frag} or die "Error, no info for frag: $frag"; + + $frag_info =~ /\[(\d+)-(\d+)\]$/ or die "Error, cannot parse coordinate info from $frag_info"; + my $frag_lend = $1; + my $frag_rend = $2; + + if (&contains_pt($frag_lend, $frag_rend, $window_lend) && &contains_pt($frag_lend, $frag_rend, $window_rend)) { + die "Error, frag $frag_info spans both edges of window[$window_lend-$window_rend] but was classified as single. "; + } + elsif (&contains_pt($frag_lend, $frag_rend, $window_lend) ) { + push (@left_single, $frag_info); + } + elsif (&contains_pt($frag_lend, $frag_rend, $window_rend) ) { + push (@right_single, $frag_info); + } + else { + die "Error, single frag: $frag_info fails to overlap either edge of the window($window_lend-$window_rend)"; + } + + } + + my @paired_frags; + foreach my $frag (@$paired_aref) { + my $frag_info = $frag_info_href->{$frag} or die "Error, no info for frag: $frag"; + + $frag_info =~ /\[(\d+)-(\d+)\]$/ or die "Error, cannot parse coordinate info from $frag_info"; + my $frag_lend = $1; + my $frag_rend = $2; + + unless (&contains_pt($frag_lend, $frag_rend, $window_lend) && &contains_pt($frag_lend, $frag_rend, $window_rend)) { + die "Error, paired frag $frag_info does NOT span both edges of window[$window_lend-$window_rend] but was classified as paired. "; + } + + push (@paired_frags, $frag_info); + } + + + + + print "\tWINDOW:$window_lend-$window_rend" + . "\tLEFT_SINGLE: " . join(",", @left_single) + . "\tRIGHT_SINGLE: " . join(",", @right_single) + . "\tPAIRS: " . join(",", @paired_frags); + + + return; + +} + + +#### +sub contains_pt { + my ($lend, $rend, $pt) = @_; + + if ($lend <= $pt && $pt <= $rend) { + return(1); + } + else { + return(0); + } +} + + + +############################################################################## +############################################################################## + +package Read_reader; + +use strict; +use warnings; +use Carp; + +sub new { + my $packagename = shift; + my ($file) = @_; + + my $self = { file => $file, + fh => undef, + line => undef, + }; + + bless ($self, $packagename); + + $self->_init(); + + return($self); +} + + +#### +sub _init { + my $self = shift; + + open (my $fh, $self->{file}) or die "Error, cannot open file: $self->{file}"; + $self->{fh} = $fh; + + $self->{line} = <$fh>; + + return; +} + +#### +sub advance_get_accessions { + my $self = shift; + my ($scaffold, $position, $type) = @_; + + + #print STDERR "advance_get_accessions($scaffold, $position, $type)\n"; + + unless ($type =~ /^LEND|REND$/) { + confess "don't understand type: $type"; + } + + my @accs; + while (1) { + unless ($self->get_curr_line()) { + return(@accs); + } + if ($scaffold ne $self->get_curr_scaffold()) { + return(@accs); + } + + my $acc = $self->get_curr_read_acc(); + my $lend = $self->get_curr_read_lend(); + my $rend = $self->get_curr_read_rend(); + + my $pos = ($type eq "LEND") ? $lend : $rend; + + + #print "$type\t$pos\n"; + + if ($pos <= $position) { + + #unless ($acc =~ m|/[12]$|) { # ignore individual reads, only consider complete fragments + push (@accs, [$acc, $lend, $rend]); + #} + + $self->advance_line(); + } + else { + return(@accs); + } + + + } + + croak "shouldn't ever get here."; + +} + + +#### +sub get_curr_line { + my $self = shift; + + return($self->{line}); +} + +#### +sub get_curr_fields { + my $self = shift; + + my $line = $self->get_curr_line(); + chomp $line; + my @x = split(/\t/, $line); + return(@x); +} + +#### +sub get_curr_scaffold { + my $self = shift; + my @x = $self->get_curr_fields(); + + return($x[0]); +} + +sub advance_to_scaffold { + my $self = shift; + my ($scaff) = @_; + + my $curr_scaff = $self->get_curr_scaffold(); + while (defined($curr_scaff) && $curr_scaff ne $scaff) { + $self->advance_line(); + $curr_scaff = $self->get_curr_scaffold(); + } + + return; +} + + +#### +sub get_curr_read_acc { + my $self = shift; + + my @x = $self->get_curr_fields(); + + return($x[1]); +} + + +#### +sub get_curr_read_lend { + my $self = shift; + my @x = $self->get_curr_fields(); + return($x[2]); +} + +#### +sub get_curr_read_rend { + my $self = shift; + my @x = $self->get_curr_fields(); + return($x[3]); +} + +#### +sub advance_line { + my $self = shift; + my $fh = $self->{fh}; + my $next_line = <$fh>; + + while ($next_line && $next_line =~ m|/[12]$|) { + $next_line = <$fh>; + } + + + $self->{line} = $next_line; + + return($next_line); +} + + diff --git a/99.scripts/trinity_utils/util/support_scripts/outfmt6_add_percent_match_length.pl b/99.scripts/trinity_utils/util/support_scripts/outfmt6_add_percent_match_length.pl new file mode 100644 index 0000000..21aaf21 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/outfmt6_add_percent_match_length.pl @@ -0,0 +1,69 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use Fasta_reader; +use List::Util qw(min max); + +my $usage = "\n\n\tusage: $0 blast.outfmt6 query_fasta target_fasta\n"; + +my $blast_file = $ARGV[0] or die $usage; +my $query_fasta = $ARGV[1] or die $usage; +my $target_fasta = $ARGV[2] or die $usage; + +main: { + + my %query_seq_lens = &get_seq_lengths($query_fasta); + + my %target_seq_lens; + if ($query_fasta eq $target_fasta) { + %target_seq_lens = %query_seq_lens; + } + else { + %target_seq_lens = &get_seq_lengths($target_fasta); + } + + open (my $fh, $blast_file) or die "Error, cannot open file $blast_file"; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my $query_acc = $x[0]; + my $target_acc = $x[1]; + + my $query_len = $query_seq_lens{$query_acc} or die "Error, cannot find seq length for query: $query_acc"; + my $target_len = $target_seq_lens{$target_acc} or die "Error, cannot find seq length for target: $target_acc"; + + my $query_hit_len = abs($x[7]-$x[6]); + my $db_hit_len = abs($x[9]-$x[8]); + + my $pct_query_len = sprintf("%.2f", $query_hit_len / $query_len * 100); + my $pct_target_len = sprintf("%.2f", $db_hit_len / $target_len * 100); + + push (@x, $query_len, $pct_query_len, $target_len, $pct_target_len, max($pct_query_len, $pct_target_len)); + + print join("\t", @x) . "\n"; + } + + exit(0); +} + +#### +sub get_seq_lengths { + my ($fasta_file) = @_; + + my %seq_lens; + + my $fasta_reader = new Fasta_reader($fasta_file); + while (my $seq_obj = $fasta_reader->next()) { + + my $acc = $seq_obj->get_accession(); + my $seq_len = length($seq_obj->get_sequence()); + + $seq_lens{$acc} = $seq_len; + } + + return(%seq_lens); +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/partition_chrysalis_graphs_n_reads.pl b/99.scripts/trinity_utils/util/support_scripts/partition_chrysalis_graphs_n_reads.pl new file mode 100644 index 0000000..2ac78d8 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/partition_chrysalis_graphs_n_reads.pl @@ -0,0 +1,250 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use File::Basename; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + +my $usage = <<__EOUSAGE__; + +######################################################################### +# +# --deBruijns bundled_iworm_contigs.fasta.deBruijn +# --componentReads readsToComponents.out.sort +# +# -N number of graphs per partition +# -L min contig length +# +########################################################################## + +__EOUSAGE__ + + + ; + + + +my $help_flag; +my $deBruijns_file; +my $componentReads_file; +my $num_graphs_per_partition; +my $min_contig_length; +my $components_directory; + +&GetOptions ( 'h' => \$help_flag, + 'deBruijns=s' => \$deBruijns_file, + 'componentReads=s' => \$componentReads_file, + 'N=s' => \$num_graphs_per_partition, + 'L=s' => \$min_contig_length, + 'compdir=s' => \$components_directory + + ); + + +if ($help_flag) { + die $usage; +} + +unless ($deBruijns_file && $componentReads_file && $num_graphs_per_partition && $min_contig_length) { + die $usage; +} + +main: { + + print STDERR "Partitioning chrysalis graphs and reads\n"; + my $outdir_base = dirname($deBruijns_file); + my $components_directory = $outdir_base . '/Component_bins'; + + my %component_id_to_partition_base; + + my $component_reader = Component_reader->new($deBruijns_file); + + my $num_components = 0; + + unless (-e $components_directory) { + mkdir $components_directory or die "Error, cannot mkdir $components_directory"; + } + + while (my $component = $component_reader->next_component()) { + + my $id = $component->{component_id}; + my $num_kmers = $component->{num_kmers}; + my $graph_text = $component->{graph_text}; + + #print join("\t", $id, $num_kmers) . "\n"; + + if ($num_kmers + 24 < $min_contig_length) { # assuming the 25-mer kmer size + next; + } + + $num_components++; + my $outdir = "$components_directory/Cbin" . int($num_components/$num_graphs_per_partition); + unless (-d $outdir) { + mkdir $outdir or die "Error, cannot mkdir $outdir"; + } + my $component_file = "$outdir/c$id.graph.tmp"; + print STDERR "\r[$num_components] writing to $component_file " if $num_components % 1000 == 0; + open (my $ofh, ">$component_file") or die "Error, cannot write to $component_file"; + print $ofh $graph_text; + close $ofh; + + $component_id_to_partition_base{$id} = "$outdir/c$id"; + + } + + print STDERR "\n\nDone partitioning graphs.\nPartitioning reads...\n\n"; + + ## Now, partition the reads + my $prev_comp = -1; + open (my $fh, "$componentReads_file") or die $!; + my $outfh; + while (my $line = <$fh>) { + chomp $line; + my ($comp, $acc, $pct, $read) = split(/\t/, $line); + if ($comp != $prev_comp) { + ## see if we need to capture these reads + $outfh = undef; + if (my $base = $component_id_to_partition_base{$comp}) { + my $outfile = $base . ".reads.tmp"; + open ($outfh, ">$outfile") or die "Error, cannot write to $outfile"; + } + } + if ($outfh) { + print $outfh "$acc $pct\n$read\n"; + } + $prev_comp = $comp; + } + close $outfh if $outfh; + close $fh; + + print STDERR "Done partitioning reads.\n\n"; + + ## identify components to target for quantifygraph: + { + my $list_output = $outdir_base.'/component_base_listing.txt'; + print STDERR "-writing $list_output\n"; + + open (my $ofh, ">$list_output") or die $!; + foreach my $component_id (sort {$a<=>$b} keys %component_id_to_partition_base) { + + my $base = $component_id_to_partition_base{$component_id}; + my $tmp_graph_file = $base . ".graph.tmp"; + my $tmp_reads_file = $base . ".reads.tmp"; + if (-s $tmp_graph_file && -s $tmp_reads_file) { + + print $ofh join("\t", $component_id, $base) . "\n"; + } + else { + print STDERR "Warning: no reads written for entry: $tmp_graph_file\n"; + } + } + close $ofh; + + print STDERR "\nDone.\n\n"; + } + + exit(0); +} + + +########################## +########################## +package Component_reader; + +use strict; +use warnings; +use Carp; + +sub new { + my ($packagename) = shift; + my ($filename) = @_; + + my $self = { filename => $filename, + fh => undef, + prev_line => undef, + }; + + bless ($self, $packagename); + + $self->_init(); + + return($self); +} + +#### +sub _init { + my $self = shift; + + open (my $fh, $self->{filename}) or die "Error, cannot open file $self-><{filename}"; + $self->{fh} = $fh; + $self->{prev_line} = <$fh>; + + unless ($self->{prev_line} =~ /^Component/) { + confess "Error, did not extract component line from file: $self->{filename}"; + } + + return; + + +} + + +#### +sub next_component { + my $self = shift; + + my @lines; + my $fh = $self->{fh}; + + my $component_id; + my $component_line = $self->{prev_line}; + if ($component_line =~ /Component (\d+)/) { + $component_id = $1; + } + + + while (my $line = <$fh>) { + if ($line =~ /^Component/) { + $self->{prev_line} = $line; + last; + } + else { + push (@lines, $line); + } + } + if (@lines) { + my $num_kmers = scalar(@lines); + my $graph_text = join("", $component_line, @lines); # newlines already included + my $component = Component->new($component_id, $num_kmers, $graph_text); + return($component); + } + else { + return(undef); + } +} + + + +#################### +#################### +package Component; + +use strict; +use warnings; +use Carp; + +sub new { + my $packagename = shift; + my ($component_id, $num_kmers, $graph_text) = @_; + + my $self = { component_id => $component_id, + num_kmers => $num_kmers, + graph_text => $graph_text, + }; + + bless ($self, $packagename); + + return($self); +} + + diff --git a/99.scripts/trinity_utils/util/support_scripts/partitioned_trinity_aggregator.pl b/99.scripts/trinity_utils/util/support_scripts/partitioned_trinity_aggregator.pl new file mode 100644 index 0000000..65f6b3e --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/partitioned_trinity_aggregator.pl @@ -0,0 +1,155 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use Getopt::Long qw(:config posix_default no_ignore_case bundling pass_through); + +my $help_flag; + + +my $usage = <<__EOUSAGE; + +################################################# +# +# usage: $0 [opts] < trinity_fasta_files_listing +# +################################################# +# +# Required +# +# --token_prefix eg. TRINITY_DN or TRINITY_GG +# +# --output_prefix eg. Trinity or Trinity_GG +# +# Optional: +# +# --include_supertranscripts flag, captures the supertranscripts fasta and gtf +# +################################################## + + + + +__EOUSAGE + + ; + + +my $token_prefix; +my $output_prefix; +my $include_supertranscripts_flag; + +&GetOptions ( 'h' => \$help_flag, + 'token_prefix=s' => \$token_prefix, + 'output_prefix=s' => \$output_prefix, + 'include_supertranscripts' => \$include_supertranscripts_flag, + ); + + +if ($help_flag) { + die $usage; +} + +unless ($token_prefix && $output_prefix) { + die $usage; +} + +my $ofh_trinity_fasta; +open($ofh_trinity_fasta, ">$output_prefix.fasta") or die "Error, cannot write to $output_prefix.fasta"; + +my $ofh_supertrans_fasta; +my $ofh_supertrans_gtf; +if ($include_supertranscripts_flag) { + open($ofh_supertrans_fasta, ">$output_prefix.SuperTrans.fasta") or die $!; + open($ofh_supertrans_gtf, ">$output_prefix.SuperTrans.gtf") or die $!; +} + + +main: { + + while () { + my $filename = $_; + chomp $filename; + unless (-e $filename) { + print STDERR "ERROR, filename: $filename is indicated to not exist.\n"; + next; + } + if (-s $filename) { + + # ie. read_partitions/Fb_0/CBin_161/c16167.trinity.reads.fa + + $filename =~ m|c(\d+)\.trinity\.reads| or die "Error, cannot parse Trinity component value from filename: $filename"; + my $component = $1; + + &process_files($filename, $component); + + } + } + + exit(0); +} + + +#### +sub process_files { + my ($filename, $component) = @_; + + { + # write fasta + open (my $fh, $filename) or die "Error, cannot open file $filename"; + while (<$fh>) { + if (/>/) { + s/>/>${token_prefix}${component}\_/; + } + print $ofh_trinity_fasta $_; + } + close $fh; + } + + if ($include_supertranscripts_flag) { + # write supertranscripts + + my $core_filename = $filename; + $core_filename =~ s/\.fasta$//; + + { # supertranscripts fasta: + my $supertranscripts_fasta = "$core_filename.SuperTrans.fasta"; + if (! -s $supertranscripts_fasta) { + die "Error, missing file: $supertranscripts_fasta"; + } + open(my $fh, $supertranscripts_fasta) or die "Error, cannot open file: $supertranscripts_fasta"; + while (<$fh>) { + if (/>/) { + s/>/>${token_prefix}${component}\_/; + } + print $ofh_supertrans_fasta $_; + } + close $fh; + } + + { + # supertranscripts gtf + my $supertranscripts_gtf = "$core_filename.SuperTrans.gtf"; + if (! -s $supertranscripts_gtf) { + die "Error, missing file: $supertranscripts_gtf"; + } + open(my $fh, $supertranscripts_gtf) or die "Error, cannot open file $supertranscripts_gtf"; + while (<$fh>) { + my $line = $_; + if (/\w/) { + + my @x = split(/\t/, $line); + my $id = $x[0]; + $line =~ s/$id/${token_prefix}${component}_${id}/g; + + } + print $ofh_supertrans_gtf $line; + } + close $fh; + } + } + + return; +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/plugin_install_tests.sh b/99.scripts/trinity_utils/util/support_scripts/plugin_install_tests.sh new file mode 100644 index 0000000..886922c --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/plugin_install_tests.sh @@ -0,0 +1,17 @@ +#!/bin/bash + +echo "## Checking plugin installations:" +echo + +if [ -e "trinity-plugins/slclust/bin/slclust" ] +then + echo "slclust: has been Installed Properly" +else + echo "slclust Installation appears to have FAILED" +fi +if [ -e "trinity-plugins/COLLECTL/collectl/collectl" ] +then + echo "collectl: has been Installed Properly" +else + echo "collectl Installation appears to have FAILED" +fi diff --git a/99.scripts/trinity_utils/util/support_scripts/prep_rnaseq_alignments_for_genome_assisted_assembly.pl b/99.scripts/trinity_utils/util/support_scripts/prep_rnaseq_alignments_for_genome_assisted_assembly.pl new file mode 100644 index 0000000..e96b9fb --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/prep_rnaseq_alignments_for_genome_assisted_assembly.pl @@ -0,0 +1,226 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use Carp; +use threads; + +use File::Basename; +use File::Spec; +use FindBin; +use Getopt::Long qw(:config no_ignore_case bundling); +use lib("$FindBin::Bin/../../PerlLib"); +use Thread_helper; +use Cwd; + +my $CPU = 2; + +my $usage = <<_EOUSAGE_; + +######################################################################################################## +# +# Required: +# +# --coord_sorted_SAM coordinate-sorted SAM file. +# +# -I maximum intron length +# (reads with longer intron lengths are ignored, and fragment reads +# farther apart on the genome are treated as unpaired)) +# -C min coverage for region boundary (default: 1) +# +# *If Strand-specific, specify: +# --SS_lib_type library type: if single: F or R, if paired: FR or RF +# +# +# Optional: +# +# --min_reads_per_partition default: 10 +# --parts_per_directory default: 100 +# +# --sort_buffer default: '10G' amount of RAM to allocate to sorting. +# +# --CPU number of threads +# +######################################################################################################## + + +_EOUSAGE_ + + ; + +# -J region join length (neighboring coverage bins within this range are merged into larger piles) + + + +my $help_flag; + +#my $partition_join_size; +my $max_intron_length; +my $SAM_file; +my $SS_lib_type = ""; +my $min_coverage = 1; +my $min_reads_per_partition = 10; +my $parts_per_dir = 100; +my $sort_buffer = '10G'; + +my $SYMLINK = ($ENV{NO_SYMLINK}) ? "cp" : "ln -sf"; + +&GetOptions ( 'h' => \$help_flag, + + 'coord_sorted_SAM=s' => \$SAM_file, + 'SS_lib_type=s' => \$SS_lib_type, + + #'J=i' => \$partition_join_size, + 'I=i' => \$max_intron_length, + 'C=i' => \$min_coverage, + + 'min_reads_per_partition=i' => \$min_reads_per_partition, + 'parts_per_directory=i' => \$parts_per_dir, + 'CPU=i' => \$CPU, + + 'sort_buffer=s' => \$sort_buffer, + ); + + +if ($help_flag) { + die $usage; +} + +unless ( + $SAM_file + # && $partition_join_size + && $max_intron_length + ) { + die $usage; +} + +if ($SS_lib_type && $SS_lib_type !~ /^(F|R|FR|RF)$/) { + die "Error, invalid --SS_lib_type, only F, R, FR, or RF are possible values"; +} + +my $UTIL_DIR = "$FindBin::RealBin/"; + +main: { + + my @sam_info; + + if ($SS_lib_type) { + my $sam_basename = basename($SAM_file); + my ($plus_strand_sam, $minus_strand_sam) = ("$sam_basename.+.sam", "$sam_basename.-.sam"); + if (-s $plus_strand_sam && $minus_strand_sam) { + print STDERR "-strand partitioned SAM files already exist, so using them instead of re-creating them.\n"; + } + else { + my $cmd = "$UTIL_DIR/SAM_strand_separator.pl $SAM_file $SS_lib_type"; + &process_cmd($cmd); + } + + push (@sam_info, [$plus_strand_sam, '+'], [$minus_strand_sam, '-']); + } + else { + if (! -e basename($SAM_file) ) { #cwd() ne dirname(File::Spec->rel2abs($SAM_file))) { + &process_cmd("$SYMLINK $SAM_file " . basename($SAM_file)); + } + $SAM_file = basename($SAM_file); + push (@sam_info, [$SAM_file, '+']); + } + + + my $thread_helper = new Thread_helper($CPU); + + foreach my $sam_info_aref (@sam_info) { + + my ($sam, $strand) = @$sam_info_aref; + + my $thread = threads->create('prep_read_partitions', $sam, $strand); + #push (@threads, $thread); + + $thread_helper->add_thread($thread); + + } + + $thread_helper->wait_for_all_threads_to_complete(); + + my @failures = $thread_helper->get_failed_threads(); + my $ret = 0; + if (@failures) { + foreach my $thread (@failures) { + if (my $error = $thread->error()) { + print STDERR "Error, thread exited with error $error\n"; + $ret++; + } + } + } + + print "##\nDone\n##\n\n" unless($ret); + + exit($ret); + + + + +} + + +sub prep_read_partitions { + my ($sam, $strand) = @_; + + ## define fragments + my $cmd = "$UTIL_DIR/SAM_to_frag_coords.pl --CPU $CPU --sort_buffer $sort_buffer --sam $sam --min_insert_size 1 --max_insert_size $max_intron_length "; ## writes file: $sam_file.frag_coords + &process_cmd($cmd) unless (-s "$sam.frag_coords"); + + ## define coverage + $cmd = "$UTIL_DIR/fragment_coverage_writer.pl $sam.frag_coords > $sam.frag_coverage.wig"; + + unless (-s "$sam.frag_coverage.wig.ok") { + &process_cmd($cmd); + &process_cmd("touch $sam.frag_coverage.wig.ok"); + } + + my $partitions_file = "$sam.minC$min_coverage.gff"; + + ## define partitions + $cmd = "$UTIL_DIR/define_coverage_partitions.pl $sam.frag_coverage.wig $min_coverage $strand > $partitions_file"; + unless (-s "$partitions_file.ok") { + &process_cmd($cmd); + &process_cmd("touch $partitions_file.ok"); + } + + + ## extract reads per partition + $cmd = "$UTIL_DIR/extract_reads_per_partition.pl --partitions_gff $partitions_file " + . " --coord_sorted_SAM $sam" + . " --parts_per_directory $parts_per_dir" + . " --min_reads_per_partition $min_reads_per_partition "; + + if ($SS_lib_type) { + $cmd .= " --SS_lib_type $SS_lib_type "; + } + + my $partitions_dir = "Dir_" . basename($partitions_file); + unless (-d $partitions_dir && -e "$partitions_dir.ok") { + &process_cmd($cmd); + &process_cmd("touch $partitions_dir.ok"); + } + + return; + + +} + + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + + my $ret = system($cmd); + + if ($ret) { + confess "Error, command $cmd died with ret $ret"; + } + + return; +} diff --git a/99.scripts/trinity_utils/util/support_scripts/print_butterfly_assemblies.pl b/99.scripts/trinity_utils/util/support_scripts/print_butterfly_assemblies.pl new file mode 100644 index 0000000..7433ab1 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/print_butterfly_assemblies.pl @@ -0,0 +1,56 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use Fasta_reader; + +my $usage = "usage: $0 chrysalis_component_listing.txt min_seq_length\n\n"; + +my $comp_list_file = $ARGV[0] or die $usage; +my $min_seq_length = $ARGV[1] or die $usage; + + +main: { + + my $ret_val = 0; + + my %seen; + + open (my $fh, $comp_list_file) or die "Error, cannot open file $comp_list_file"; + while (<$fh>) { + chomp; + my ($comp_id, $comp_base) = split(/\t/); + my $butterfly_fasta_file = "$comp_base.graph.allProbPaths.fasta"; + + if (-e $butterfly_fasta_file) { + + my $fasta_reader = new Fasta_reader($butterfly_fasta_file); + while (my $seq_obj = $fasta_reader->next()) { + my $sequence = $seq_obj->get_sequence(); + if (length($sequence) >= $min_seq_length) { + + if ($seen{$sequence}) { + print STDERR "-duplicate sequence detected, excluding it.\n"; + next; + } + else { + $seen{$sequence} = 1; + } + + print $seq_obj->get_FASTA_format(); + } + } + } + else { + print STDERR "Error, no fasta file reported as: $butterfly_fasta_file\n"; + $ret_val = 1; + } + } + + close $fh; + + exit($ret_val); +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/process_GMAP_alignments_gff3_chimeras_ok.pl b/99.scripts/trinity_utils/util/support_scripts/process_GMAP_alignments_gff3_chimeras_ok.pl new file mode 100644 index 0000000..7282fcd --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/process_GMAP_alignments_gff3_chimeras_ok.pl @@ -0,0 +1,95 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use File::Basename; +use Cwd; + +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + + +my $usage = <<__EOUSAGE__; + +###################################################################### +# +# Required: +# --genome target genome to align to +# --transcripts cdna sequences to align +# +# Optional: +# -I max intron length +# --CPU number of threads (default: 2) +# +####################################################################### + + +__EOUSAGE__ + + ; + + +my ($genome, $transcriptDB, $max_intron); +my $CPU = 2; + +my $help_flag; + +&GetOptions( 'h' => \$help_flag, + 'genome=s' => \$genome, + 'transcripts=s' => \$transcriptDB, + 'I=i' => \$max_intron, + 'CPU=i' => \$CPU, + ); + + +unless ($genome && $transcriptDB) { + die $usage; +} + + +main: { + + my $genomeName = basename($genome); + my $genomeDir = $genomeName . ".gmap"; + + my $genomeBaseDir = dirname($genome); + + my $cwd = cwd(); + + unless (-d "$genomeBaseDir/$genomeDir") { + + my $cmd = "gmap_build -D $genomeBaseDir -d $genomeDir -k 13 $genome >&2"; + &process_cmd($cmd); + } + + + ## run GMAP + my $cmd = "gmap -D $genomeBaseDir -d $genomeDir $transcriptDB -f 3 -n 0 -x 50 -t $CPU "; + if ($max_intron) { + $cmd .= " --intronlength=$max_intron "; + } + + &process_cmd($cmd); + + exit(0); +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + #return; + + my $ret = system($cmd); + if ($ret) { + die "Error, cmd: $cmd died with ret ($ret)"; + } + + return; +} + + + diff --git a/99.scripts/trinity_utils/util/support_scripts/revcomp_fasta.pl b/99.scripts/trinity_utils/util/support_scripts/revcomp_fasta.pl new file mode 100644 index 0000000..4103487 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/revcomp_fasta.pl @@ -0,0 +1,21 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +local $/ = ">"; +while(<>){ + chomp; + my $lineSepPos = index($_, "\n"); + my $header = substr($_,0,$lineSepPos); + if($header){ + print(">", $header, "\n"); + my $sequence = reverse(substr($_,$lineSepPos)); + # see http://shootout.alioth.debian.org/u32/performance.php?test=revcomp#about + # for ambiguity codes and translation + $sequence =~ tr/ACGTUMRWSYKVHDBNacgtumrwsykvhdbn\n/TGCAAKYWSRMBDHVNtgcaakywsrmbdhvn/d; + for(my $pos = 0; $pos < length($sequence);$pos += 60){ + print(substr($sequence, $pos, 60),"\n"); + } + } +} diff --git a/99.scripts/trinity_utils/util/support_scripts/run_TMM_scale_matrix.pl b/99.scripts/trinity_utils/util/support_scripts/run_TMM_scale_matrix.pl new file mode 100644 index 0000000..8025367 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/run_TMM_scale_matrix.pl @@ -0,0 +1,207 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling); +use Cwd; +use FindBin; +use File::Basename; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Data::Dumper; + +my $usage = <<__EOUSAGE__; + + + +################################################################################################# +# +# Required: +# +# --matrix matrix of raw read counts (not normalized!) +# +################################################################################################ + + + + +__EOUSAGE__ + + + ; + + + +my $matrix_file; +my $help_flag; + +&GetOptions ( 'h' => \$help_flag, + 'matrix=s' => \$matrix_file, + ); + + +if ($help_flag) { + die $usage; +} + + +unless ($matrix_file) { + die $usage; +} + + + +main: { + + my $tmm_info_file = &run_TMM($matrix_file); + + &write_normalized_file($matrix_file, $tmm_info_file); + + exit(0); +} + + +#### +sub run_TMM { + my ($counts_matrix_file) = @_; + + my $tmm_norm_script = "$counts_matrix_file.runTMM.R"; + open (my $ofh, ">$tmm_norm_script") or die "Error, cannot write to $tmm_norm_script"; + #print $ofh "source(\"$FindBin::RealBin/R/edgeR_funcs.R\")\n"; + + print $ofh "library(edgeR)\n\n"; + + print $ofh "rnaseqMatrix = read.table(\"$counts_matrix_file\", header=T, row.names=1, com='', check.names=F)\n"; + print $ofh "rnaseqMatrix = as.matrix(rnaseqMatrix)\n"; + print $ofh "rnaseqMatrix = round(rnaseqMatrix)\n"; + print $ofh "exp_study = DGEList(counts=rnaseqMatrix, group=factor(colnames(rnaseqMatrix)))\n"; + print $ofh "exp_study = calcNormFactors(exp_study)\n"; + + print $ofh "exp_study\$samples\$eff.lib.size = exp_study\$samples\$lib.size * exp_study\$samples\$norm.factors\n"; + print $ofh "write.table(exp_study\$samples, file=\"$counts_matrix_file.TMM_info.txt\", quote=F, sep=\"\\t\", row.names=F)\n"; + + close $ofh; + + &process_cmd("R --no-save --no-restore --no-site-file --no-init-file -q < $tmm_norm_script 1>&2 "); + + my $tmm_matrix = "$counts_matrix_file.TMM_info.txt"; + + unless (-s $tmm_matrix) { + confess "Error, TMM matrix $tmm_matrix was not generated. Be sure edgeR is installed and see additional error messages above for other helpful info"; + } + + return($tmm_matrix); + +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret ($ret) "; + } + + return; +} + +#### +sub write_normalized_file { + my ($matrix_file, $tmm_info_file) = @_; + + my %col_to_eff_lib_size; + my %col_to_norm_factor; + open (my $fh, $tmm_info_file) or die "Error, cannot open file $tmm_info_file"; + + my %bloom_to_col; + + my $header = <$fh>; + while (<$fh>) { + chomp; + my @x = split(/\t/); + my ($col, $norm_factor, $eff_lib_size) = ($x[0], $x[2], $x[3]); + + $col =~ s/\"//g; + + my $bloom = $col; + $bloom =~ s/\W/$;/g; + + $col_to_eff_lib_size{$col} = $eff_lib_size; + $col_to_norm_factor{$col} = $norm_factor; + + if ($bloom ne $col) { + + if (exists $bloom_to_col{$bloom}) { + die "Error, already stored $bloom_to_col{$bloom} for $bloom, but trying to also store $col here... Ensure column names are unique according to non-word characters."; + } + + $col_to_eff_lib_size{$bloom} = $eff_lib_size; + $col_to_norm_factor{$bloom} = $norm_factor; + } + + } + close $fh; + + open ($fh, $matrix_file); + $header = <$fh>; + chomp $header; + my @pos_to_col = split(/\t/, $header); + my $check_column_ordering_flag = 0; + + while (<$fh>) { + chomp; + my @x = split(/\t/); + + unless ($check_column_ordering_flag) { + if (scalar(@x) == scalar(@pos_to_col) + 1) { + ## header is offset, as is acceptable by R + ## not acceptable here. fix it: + unshift (@pos_to_col, ""); + } + $check_column_ordering_flag = 1; + print join("\t", @pos_to_col) . "\n"; + } + + + my $gene = $x[0]; + + print $gene; + for (my $i = 1; $i <= $#x; $i++) { + my $col = $pos_to_col[$i]; + + $col =~ s/\"//g; + + my $adj_col = $col; + $adj_col =~ s/-/\./g; + + my $bloom = $col; + $bloom =~ s/\W/$;/g; + + my $eff_lib_size = $col_to_eff_lib_size{$col} + || $col_to_eff_lib_size{$bloom} + || $col_to_eff_lib_size{$adj_col} + || $col_to_eff_lib_size{"X$col"} + || die "Error, no eff lib size for [$col] or [$bloom] or [$adj_col] or [\"X$col\"]" . Dumper(\%col_to_eff_lib_size); + + my $norm_factor = $col_to_norm_factor{$col} + || $col_to_norm_factor{$bloom} + || $col_to_norm_factor{$adj_col} + || $col_to_norm_factor{"X$col"} + || die "Error, no normalization scaling factor for $col" . Dumper(\%col_to_norm_factor); + + + my $read_count = $x[$i]; + + my $converted_val = sprintf("%.3f", $read_count * 1/$norm_factor); + + print "\t$converted_val"; + } + print "\n"; + } + + return; + +} diff --git a/99.scripts/trinity_utils/util/support_scripts/run_UpperQuartileNormalization_matrix.Rscript b/99.scripts/trinity_utils/util/support_scripts/run_UpperQuartileNormalization_matrix.Rscript new file mode 100644 index 0000000..67c187d --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/run_UpperQuartileNormalization_matrix.Rscript @@ -0,0 +1,35 @@ +#!/usr/bin/env Rscript + +################################################ +# for info on upper quartile normalization, see: +# http://vinaykmittal.blogspot.com/2013/10/fpkmrpkm-normalization-caveat-and-upper.html +################################################ + +suppressPackageStartupMessages(library("argparse")) + +parser = ArgumentParser() +parser$add_argument("--matrix", help="input data file", required=TRUE, nargs=1) +parser$add_argument("--output", help="output data file", required=TRUE, nargs=1) + +args = parser$parse_args() +matrix_filename = args$matrix +output_filename = args$output + +data = read.table(gzfile(matrix_filename), header=T, sep="\t", row.names=1, check.names=F, com='') + + +get_upper_quartile = function(vec) { + vec = vec[vec > 0] + quantile(vec, 0.75) +} + +data = as.matrix(data) +upp_quartiles = apply(data, 2, get_upper_quartile) +m = sweep(data, MARGIN=2, upp_quartiles, '/') +mean_upp_quart = mean(upp_quartiles) +m = m * mean_upp_quart +write.table(m, file=output_filename, quote=F, sep="\t", col.names=NA) + + +quit(save = "no", status = 0, runLast = FALSE) + diff --git a/99.scripts/trinity_utils/util/support_scripts/run_UpperQuartileNormalization_matrix.pl b/99.scripts/trinity_utils/util/support_scripts/run_UpperQuartileNormalization_matrix.pl new file mode 100644 index 0000000..75e9d3c --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/run_UpperQuartileNormalization_matrix.pl @@ -0,0 +1,115 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Carp; +use Getopt::Long qw(:config no_ignore_case bundling); +use Cwd; +use FindBin; +use File::Basename; +use lib ("$FindBin::RealBin/../../PerlLib"); +use Data::Dumper; + + +################################################ +# for info on upper quartile normalization, see: +# http://vinaykmittal.blogspot.com/2013/10/fpkmrpkm-normalization-caveat-and-upper.html +################################################ + +my $usage = <<__EOUSAGE__; + + + +################################################################################################################### +# +# Required: +# +# --matrix FPKM matrix +# +# --min_expr_in_calc minimum (exclusive) fpkm value to consider in computing the upper quartile of the distribution. +# default: must be > 0 +# +#################################################################################################################### + + + + +__EOUSAGE__ + + + ; + + + +my $matrix_file; +my $help_flag; +my $min_expr_in_calc = 0; + +&GetOptions ( 'h' => \$help_flag, + 'matrix=s' => \$matrix_file, + 'min_expr_in_calc=f' => \$min_expr_in_calc, + ); + + +if ($help_flag) { + die $usage; +} + + +unless ($matrix_file) { + die $usage; +} + + + +main: { + + &upper_quartile_normalize($matrix_file); + + exit(0); +} + + +#### +sub upper_quartile_normalize { + my ($matrix_file) = @_; + + my $tmm_norm_script = "__tmp_upper_quart_norm.R"; + open (my $ofh, ">$tmm_norm_script") or die "Error, cannot write to $tmm_norm_script"; + #print $ofh "source(\"$FindBin::RealBin/R/edgeR_funcs.R\")\n"; + + print $ofh "data = read.table(\"$matrix_file\", header=T, row.names=1, com='')\n"; + print $ofh "get_upper_quartile = function(vec) {\n" + . " vec = vec[vec > $min_expr_in_calc]\n" + . " quantile(vec, 0.75)\n" + . "}\n"; + + print $ofh "data = as.matrix(data)\n"; + print $ofh "upp_quartiles = apply(data, 2, get_upper_quartile)\n"; + print $ofh "m = sweep(data, MARGIN=2, upp_quartiles, '/')\n"; + print $ofh "mean_upp_quart = mean(upp_quartiles)\n"; + print $ofh "m = m * mean_upp_quart\n"; + + print $ofh "write.table(m, quote=F, sep=\"\\t\")\n"; + + close $ofh; + + &process_cmd("R --no-save --no-restore --no-site-file --no-init-file -q --slave < $tmm_norm_script "); + + return; +} + +#### +sub process_cmd { + my ($cmd) = @_; + + print STDERR "CMD: $cmd\n"; + my $ret = system($cmd); + + if ($ret) { + die "Error, cmd: $cmd died with ret ($ret) "; + } + + return; +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/salmon_runner.pl b/99.scripts/trinity_utils/util/support_scripts/salmon_runner.pl new file mode 100644 index 0000000..dfee388 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/salmon_runner.pl @@ -0,0 +1,48 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use Process_cmd; + +my $usage = "\n\n\tusage: $0 Trinity.fasta reads.fa [threads=1]\n\n"; + +my $trin_fa = $ARGV[0] or die $usage; +my $reads_fa = $ARGV[1] or die $usage; +my $CPU = $ARGV[2] || 1; + + + + +main: { + + my $salmon_index = "$trin_fa.salmon.idx"; + my $salmon_stderr = "_salmon.$$.stderr"; + my $cmd = "salmon --no-version-check index -t $trin_fa -i $salmon_index -k 25 -p $CPU > $salmon_stderr 2>&1"; + &run_cmd_capture_stderr($cmd, $salmon_stderr); + + + $cmd = "salmon --no-version-check quant -i $salmon_index -l U -r $reads_fa -o salmon_outdir -p $CPU --minAssignedFrags 1 --validateMappings > $salmon_stderr 2>&1"; + &run_cmd_capture_stderr($cmd, $salmon_stderr); + + unlink($salmon_stderr); + + exit(0); + +} + + +#### +sub run_cmd_capture_stderr { + my ($cmd, $stderr_file) = @_; + eval { + &process_cmd($cmd); + }; + if ($@) { + my $errmsg = `cat $stderr_file`; + die "Error, cmd: $cmd failed with msg: $errmsg $@"; + } + return; +} diff --git a/99.scripts/trinity_utils/util/support_scripts/salmon_trans_to_gene_results.pl b/99.scripts/trinity_utils/util/support_scripts/salmon_trans_to_gene_results.pl new file mode 100644 index 0000000..b776d38 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/salmon_trans_to_gene_results.pl @@ -0,0 +1,165 @@ +#!/usr/bin/env perl + +use strict; +use warnings; +use Data::Dumper; + +my $usage = "\n\nusage: $0 quant.sf gene_to_trans_map_file.txt\n\n\n"; + +my $quant_sf = $ARGV[0] or die $usage; +my $gene_to_trans_map_file = $ARGV[1] or die $usage; + + +main: { + + my %trans_to_gene_info; + { + open (my $fh, $gene_to_trans_map_file) or die "Error, cannot open file $gene_to_trans_map_file"; + while (<$fh>) { + unless (/\w/) { next; } + chomp; + my ($gene, $trans, @rest) = split(/\s+/); + unless ($gene && $trans) { + die "Error, cannot extract gene & trans relationship from line $_ of file $gene_to_trans_map_file"; + } + $trans_to_gene_info{$trans} = $gene; + } + close $fh; + } + + + open (my $fh, $quant_sf) or die "Error, cannot open file $quant_sf"; + my $header = <$fh>; + chomp $header; + my %field_index; + my @fields = split(/\t/, $header); + { + + for (my $i = 0; $i <= $#fields; $i++) { + my $field = $fields[$i]; + $field_index{$field} = $i; + } + } + + + my %gene_data; + while (<$fh>) { + chomp; + + # quant.sf format: + # + #Name Length EffectiveLength TPM NumReads + #TRINITY_DN10_c0_g1_i1 334 67.2849 3125.31 7 + #TRINITY_DN11_c0_g1_i1 319 55.1277 0 0 + #TRINITY_DN12_c0_g1_i1 244 244 1231.18 10 + #TRINITY_DN17_c0_g1_i1 229 229 393.549 3 + #TRINITY_DN18_c0_g1_i1 633 360.371 593.619 7.12107 + + my @x = split(/\t/); + + my $trans_id = $x[ $field_index{Name} ]; + my $tpm = $x[ $field_index{TPM} ]; + my $length = $x[ $field_index{Length} ]; + my $eff_length = $x[ $field_index{EffectiveLength} ]; + my $est_counts = $x[ $field_index{NumReads} ]; + + my $gene = $trans_to_gene_info{$trans_id} or die "Error, cannot find gene identifier for transcript [$trans_id] "; + + push (@{$gene_data{$gene}}, { Name => $trans_id, + TPM => $tpm, + Length => $length, + EffectiveLength => $eff_length, + NumReads => $est_counts, + }); + + + } + close $fh; + + + ## Output gene summaries: + + print $header . "\n"; + + foreach my $gene (keys %gene_data) { + my @trans_structs = @{$gene_data{$gene}}; + + my @trans_ids; + my $sum_counts = 0; + my $sum_tpm = 0; + + my $counts_per_len_sum = 0; + my $counts_per_eff_len_sum = 0; + + my $sum_lengths = 0; + my $sum_eff_lengths = 0; + + my $num_trans = scalar(@trans_structs); + + + foreach my $struct (@trans_structs) { + + #print Dumper($struct); + + my $trans_id = $struct->{Name}; + my $tpm = $struct->{TPM}; + my $length = $struct->{Length}; + + my $eff_length = $struct->{EffectiveLength}; + my $est_counts = $struct->{NumReads}; + + unless ($eff_length > 0) { + $eff_length = 1; # cannot have zero length feature! + } + + unless ($length > 0 && $eff_length > 0) { + die "Error, length: $length, eff_length: $eff_length" . Dumper($struct); + } + + $sum_lengths += $length; + $sum_eff_lengths += $eff_length; + + $counts_per_len_sum += $est_counts/$length; + + $counts_per_eff_len_sum += $est_counts/$eff_length; + + $sum_counts += $est_counts; + $sum_tpm += $tpm; + } + + my $gene_length = $sum_lengths / $num_trans; + my $gene_eff_length = $sum_eff_lengths / $num_trans; + if ($sum_counts) { + # set lengths as weighted by expression of isoforms. + eval { + $gene_length = $sum_counts / $counts_per_len_sum; + $gene_eff_length = $sum_counts / $counts_per_eff_len_sum; + }; + if ($@) { + print STDERR "$@\n" . Dumper(\@trans_structs); + die; + } + } + + my %gene_info = ( Name => $gene, + TPM => sprintf("%.2f", $sum_tpm), + Length => sprintf("%.2f", $gene_length), + EffectiveLength => sprintf("%.2f", $gene_eff_length), + NumReads => sprintf("%.2f", $sum_counts), + + ); + + my @vals; + foreach my $field (@fields) { + my $result = $gene_info{$field}; + unless (defined $result) { + $result = "NA"; + } + push (@vals, $result); + } + print join("\t", @vals) . "\n"; + } + + + exit(0); +} diff --git a/99.scripts/trinity_utils/util/support_scripts/scaffold_iworm_contigs.pl b/99.scripts/trinity_utils/util/support_scripts/scaffold_iworm_contigs.pl new file mode 100644 index 0000000..bb49892 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/scaffold_iworm_contigs.pl @@ -0,0 +1,287 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use lib ("$FindBin::Bin/../../PerlLib"); +use SAM_entry; + +my $usage = "\n\nusage: $0 nameSorted.sam iworm_fasta_file\n\n"; + +my $name_sorted_sam_file = $ARGV[0] or die $usage; +my $iworm_fasta = $ARGV[1] or die $usage; + + +my $SAM_OFH; + + +main: { + + open ($SAM_OFH, ">scaffolding_entries.sam") or die $!; + + my %iworm_acc_to_fasta_index; + { + + my $counter = 0; + open (my $fh, $iworm_fasta) or die "Error, cannot open file $iworm_fasta"; + while (<$fh>) { + if (/^>(\S+)/) { + my $acc = $1; + $iworm_acc_to_fasta_index{$acc} = $counter; # starts at zero + $counter++; + } + } + close $fh; + } + + + my %paired_iworm_contigs; + + my $prev_core_acc = ""; + my %end_to_iworm; + + my $num_warnings = 0; + + my $num_unrecognized_iworm_contig_names = 0; + + if ($name_sorted_sam_file =~ /\.bam$/) { + $name_sorted_sam_file = "samtools view $name_sorted_sam_file | "; + } + + open (my $fh, $name_sorted_sam_file) or die "Error, cannot open file $name_sorted_sam_file"; + while (<$fh>) { + my $line = $_; + chomp; + my @x = split(/\t/); + my $read_acc = $x[0]; + my $iworm_acc = $x[2]; + + unless (defined $iworm_acc) { next; } + + if ($iworm_acc eq '*') { + # unmapped read + next; + } + + unless ($iworm_acc =~ /^a\d+;\d+/) { + $num_unrecognized_iworm_contig_names++; + if ($num_unrecognized_iworm_contig_names <= 10) { + print STDERR "warning, inchworm contig ($iworm_acc) isn't recognized. Ignoring entry: [[$line]]\n"; + } + if ($num_unrecognized_iworm_contig_names == 11) { + print STDERR "warning, too many unrecognized inchworm contig names. Will report summary of counts later.\n"; + } + next; + } + + + my $core_acc; + my $frag_end; + + + if ($read_acc =~ /^(\S+)\/([12])$/) { + + $core_acc = $1; + $frag_end = $2; + } + + # capture long read mappings + elsif ($read_acc =~ /^(LR\$\|\S+)/) { + $core_acc = $1; + $frag_end = "LR"; + } + + else { + # must have mixed in a single read with the pairs... + if ($num_warnings <= 10) { + print STDERR "warning, ignoring read: $read_acc since cannot decipher if /1 or /2 of a pair.\n"; + } + elsif ($num_warnings == 11) { + print STDERR "number of read warnings exceeded 10. Turning off warning messages from here out.\n"; + } + $num_warnings++; + + next; + } + + if ($core_acc ne $prev_core_acc) { + + &examine_frags(\%end_to_iworm, \%paired_iworm_contigs); + %end_to_iworm = (); + + } + + $end_to_iworm{$frag_end}->{$iworm_acc} = $line; + + $prev_core_acc = $core_acc; + + } + close $fh; + + ## get last one + &examine_frags(\%end_to_iworm, \%paired_iworm_contigs); + + if ($num_warnings) { + print STDERR "WARNING: note there were $num_warnings reads that could not be deciphered as being /1 or /2 of a PE fragment. Hopefully, these were SE reads that should have been ignored. Otherwise, please research this further.\n\n"; + } + if ($num_unrecognized_iworm_contig_names) { + print STDERR "WARNING: note, there were $num_unrecognized_iworm_contig_names inchworm contig names in the SAM file that were ignored due to the inchworm contig accession not being recognized.\n"; + } + + foreach my $pairing (reverse sort {$paired_iworm_contigs{$a}<=>$paired_iworm_contigs{$b}} keys %paired_iworm_contigs) { + + my ($iworm_acc_A, $iworm_acc_B) = split(/\t/, $pairing); + + my $index_A = $iworm_acc_to_fasta_index{$iworm_acc_A}; + unless (defined $index_A) { + print STDERR "WARNING, no index for iworm acc: $iworm_acc_A\n"; + next; + } + my $index_B = $iworm_acc_to_fasta_index{$iworm_acc_B}; + unless (defined $index_B) { + print STDERR "WARNING, no index for iworm acc: $iworm_acc_B\n"; + next; + } + + print join("\t", $iworm_acc_A, $index_A, $iworm_acc_B, $index_B, $paired_iworm_contigs{$pairing}) . "\n"; + + } + + + close $SAM_OFH; + + exit(0); +} + +#### +sub examine_frags { + my ($end_to_iworm_href, $paired_iworm_contigs_href) = @_; + + if (exists ($end_to_iworm_href->{LR})) { + my @LR_read_names = keys %{$end_to_iworm_href->{LR}}; + + for (my $i = 0; $i < $#LR_read_names; $i++) { + for (my $j = $i + 1; $j <= $#LR_read_names; $j++) { + + my $pair = join("\t", sort ($LR_read_names[$i], $LR_read_names[$j]) ); + $paired_iworm_contigs_href->{$pair}++; + + #print STDERR "-got LR pair: $pair\n"; + } + } + + return; + } + + + ## handle the paired-end reads + + my @ends = keys %$end_to_iworm_href; + + unless (scalar @ends == 2) { + ## no pairs + return; + } + + + my @iworm_left = keys %{$end_to_iworm_href->{1}}; + my @iworm_right = keys %{$end_to_iworm_href->{2}}; + + + if (scalar(@iworm_left) == 1 && scalar(@iworm_right) == 1 + && + $iworm_left[0] ne $iworm_right[0]) { + + ## simplest case, pairs each aligning to different contigs. + + my $i_left = $iworm_left[0]; + my $i_right = $iworm_right[0]; + + my $pair = join("\t", sort ($i_left, $i_right)); + #print STDERR "Got pairing: $pair\n"; + $paired_iworm_contigs_href->{$pair}++; + + } + else { + # examine split reads + + &check_for_split_reads($end_to_iworm_href->{1}, $paired_iworm_contigs_href); + &check_for_split_reads($end_to_iworm_href->{2}, $paired_iworm_contigs_href); + + + } + + ## output all alignments for these reads. + foreach my $sam_line (values %{$end_to_iworm_href->{1}}, values %{$end_to_iworm_href->{2}}) { + print $SAM_OFH $sam_line; + } + + + return; +} + + +#### +sub check_for_split_reads { + my ($read_mappings_href, $paired_iworm_contigs_href) = @_; + + my @iworm_names = keys %$read_mappings_href; + + if (scalar(@iworm_names) < 2) { + # nothing to do here. + return; + } + + ## see if we have a read where different parts of the read are aligning to different iworm contigs. + for (my $i = 0; $i < $#iworm_names; $i++) { + my $iworm_name_i = $iworm_names[$i]; + my $align_i = new SAM_entry($read_mappings_href->{$iworm_name_i}); + + for (my $j = $i + 1; $j <= $#iworm_names; $j++) { + my $iworm_name_j = $iworm_names[$j]; + my $align_j = new SAM_entry($read_mappings_href->{$iworm_name_j}); + + if (&overlap_not_involving_containment($align_i, $align_j)) { + ## store pair + + my $pair = join("\t", sort($iworm_name_i, $iworm_name_j)); + $paired_iworm_contigs_href->{$pair}++; + + } + } + } + + return; +} + +#### +sub overlap_not_involving_containment { + my ($align_i, $align_j) = @_; + + my ($read_i_lend, $read_i_rend) = $align_i->get_read_span(); + + my ($read_j_lend, $read_j_rend) = $align_j->get_read_span(); + + #print join("\t", "$read_i_lend-$read_i_rend", "$read_j_lend-$read_j_rend") . "\n"; + + if ( ($read_i_lend <= $read_j_lend && $read_i_rend >= $read_j_rend) # i encapsulates j + || + ($read_j_lend <= $read_i_lend && $read_j_rend >= $read_i_rend) # j encapsulates i + ) { + ## containment + #print "\t* containment\n"; + return(0); + } + elsif ( $read_i_lend < $read_j_rend && $read_i_rend > $read_j_lend) { + #print "\t OVERLAP\n"; + return(1); + } + else { + #print "\tok\n"; + + return(1); + } + +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/segment_GFF_partitions.pl b/99.scripts/trinity_utils/util/support_scripts/segment_GFF_partitions.pl new file mode 100644 index 0000000..82d24c7 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/segment_GFF_partitions.pl @@ -0,0 +1,166 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + + +my $usage = "usage: $0 partitions.gff clip_pts.wig\n\n"; + +my $partitions_gff = $ARGV[0] or die $usage; +my $clip_pts_wig = $ARGV[1] or die $usage; + +my $MIN_PARTITION_SIZE = 50; + +main: { + + my %scaff_to_gff = &parse_gff_entries($partitions_gff); + my %scaff_to_clip = &parse_wig_entries($clip_pts_wig); + + + foreach my $scaff (sort keys %scaff_to_gff) { + + my @gffs = @{$scaff_to_gff{$scaff}}; + @gffs = sort {$a->{lend}<=>$b->{lend}} @gffs; + + my @clips; + if (exists $scaff_to_clip{$scaff}) { + @clips = @{$scaff_to_clip{$scaff}}; + @clips = sort {$a<=>$b} @clips; + } + + foreach my $gff (@gffs) { + + my ($lend, $rend) = ($gff->{lend}, $gff->{rend}); + + my @overlapping_clips; + + if (@clips) { + while (@clips && $clips[0] < $rend) { + my $clip = shift @clips; + if ($clip > $lend) { + push (@overlapping_clips, $clip); + } + } + } + + if (@overlapping_clips) { + my @x = @{$gff->{line}}; + + my $gff_line = join("\t", @x); + + #print "$gff_line\nHas overlapping clips: @overlapping_clips\n"; + + + my $prev_lend = $lend; + + foreach my $overlapping_clip (@overlapping_clips) { + + my $rend = $overlapping_clip - 1; + + my @y = @x; + $y[3] = $prev_lend; + $y[4] = $rend; + + my $part_length = $rend - $prev_lend + 1; + + if ($part_length >= $MIN_PARTITION_SIZE) { + + print join("\t", @y) . "\n"; + + } + + $prev_lend = $rend + 1; + + + #print "$gff_line\nWith clips: " . join("\t", @overlapping_clips) . "\n\n"; + } + ## report last partition of clip: + + $x[3] = $prev_lend; + print join("\t", @x) . "\n"; + + + #print "\n\n"; + } + + else { + + ## no clipping. Report original partition. + + print join("\t", @{$gff->{line}}) . "\n"; + } + + + } + + } + + exit(0); + +} + + + +#### +sub parse_gff_entries { + my ($gff_file) = @_; + + my %scaffold_to_gff; + + open (my $fh, $gff_file) or die "Error, cannot open file $gff_file"; + while (<$fh>) { + chomp; + + my @x = split(/\t/); + + my $scaff = $x[0]; + my $lend = $x[3]; + my $rend = $x[4]; + + my $struct = { line => [@x], + scaff => $scaff, + lend => $lend, + rend => $rend, + }; + + push (@{$scaffold_to_gff{$scaff}}, $struct); + } + close $fh; + + return(%scaffold_to_gff); +} + + +#### +sub parse_wig_entries { + my ($jaccard_file) = @_; + + my %scaffold_to_entries; + + my $curr_scaff; + + open (my $fh, $jaccard_file) or die "Error, cannot open file $jaccard_file"; + while (<$fh>) { + unless (/\w/) { next; } + chomp; + if (/^variableStep chrom=(\S+)/) { + $curr_scaff = $1; + next; + } + + my ($coord, $val) = split(/\t/); + + unless (defined($coord) && defined($val)) { + #die "Error, line $_ lacks expected format"; + next; + } + + push (@{$scaffold_to_entries{$curr_scaff}}, $coord); + } + + close $fh; + + + return(%scaffold_to_entries); +} + diff --git a/99.scripts/trinity_utils/util/support_scripts/tests/sample_data_tests.py b/99.scripts/trinity_utils/util/support_scripts/tests/sample_data_tests.py new file mode 100644 index 0000000..df9d012 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/tests/sample_data_tests.py @@ -0,0 +1,59 @@ +from Bio import SeqIO +import unittest +import os + +class TestTrinitySampleData(unittest.TestCase): + + @classmethod + def setUpClass(cls): + cls.sampledata_dir = os.environ["TRINITY_SAMPLEDATA"] + + def test_genome_guided(self): + seq_count = self.count_sequences('test_GenomeGuidedTrinity', 'test_GG_use_bam_trinity_outdir', + 'Trinity-GG.fasta') + self.assertTrue(50 <= seq_count <= 60, msg='Found %s sequences' % seq_count) + + def test_genome_guided_with_jaccard_clipping(self): + seq_count = self.count_sequences('test_GenomeGuidedTrinity', 'test_Schizo_trinityGG_jaccard_RF_outdir', + 'Trinity-GG.fasta') + self.assertTrue(60 <= seq_count <= 80, msg='Found %s sequences' % seq_count) + + def test_paired_end_normalization(self): + seq_count = self.count_sequences('test_InSilicoReadNormalization', + 'reads.left.fq.gz.normalized_K25_C5_pctSD200.fq') + self.assertTrue(35 <= seq_count <= 50, msg='Found %s sequences' % seq_count) + seq_count = self.count_sequences('test_InSilicoReadNormalization', + 'reads.right.fq.gz.normalized_K25_C5_pctSD200.fq') + self.assertTrue(30 <= seq_count <= 40, msg='Found %s sequences' % seq_count) + seq_count = self.count_sequences('test_InSilicoReadNormalization', + 'reads.single.fq.normalized_K25_C5_pctSD200.fq') + self.assertTrue(50 <= seq_count <= 65, msg='Found %s sequences' % seq_count) + + def test_trinity_assembly(self): + seq_count = self.count_sequences('test_Trinity_Assembly', 'trinity_out_dir', 'Trinity.fasta') + self.assertTrue(100 <= seq_count <= 120, msg='Found %s sequences' % seq_count) + + def test_DE_analysis_EdgeR(self): + check_file = os.path.join(self.sampledata_dir, 'test_DE_analysis', 'edgeR_outdir', 'numDE_feature_counts.P0.001_C2.matrix') + self.assertTrue(os.path.isfile(check_file), 'DE output file not created') + + def test_align_and_estimate_abundance(self): + cats = ['PAIRED', 'SINGLE'] + for cat in cats: + check_file = os.path.join(self.sampledata_dir, 'test_align_and_estimate_abundance',\ + '%s_END_ABUNDANCE_ESTIMATION' % cat, 'RSEM-gene.counts.matrix') + self.assertTrue(os.path.isfile(check_file), 'RSEM-gene.counts.matrix not created') + + def test_full_edgeR_pipeline(self): + check_file = os.path.join(self.sampledata_dir, 'test_full_edgeR_pipeline', 'read_content_analysis', 'read_content_analysis.nameSorted.bam') + self.assertTrue(os.path.isfile(check_file), 'edgeR sorted BAM not created') + + +### Helper methods + def count_sequences(self, *paths): + gg = os.path.join(self.sampledata_dir, *paths) + handle = open(gg, "rU") + seq_count = len([x for x in SeqIO.parse(handle, "fasta")]) + handle.close() + return seq_count + diff --git a/99.scripts/trinity_utils/util/support_scripts/tests/test_prep.py b/99.scripts/trinity_utils/util/support_scripts/tests/test_prep.py new file mode 100644 index 0000000..e5b1016 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/tests/test_prep.py @@ -0,0 +1,137 @@ +import unittest +import shutil +import os +import subprocess +from Bio import SeqIO + +# single, leftright +# fasta, fastq +# one file, two files +# forward, reverse +# clear, gzip, bzip + + +class TestTrinityPrepFlag(unittest.TestCase): + + @classmethod + def setUpClass(cls): + try: + os.remove('coverage.log') + except: + pass + + def tearDown(self): + shutil.rmtree('trinity_out_dir', True) + #pass + + def test_fastq(self): + self.trinity("left1.fq", "fq") + self.assertEquals(30575, self.count_seqs(), "Unexpected sequence count") + + def test_fastq_gz(self): + self.trinity("left1.fq.gz", "fq") + self.assertEquals(30575, self.count_seqs(), "Unexpected sequence count") + + def test_fastq_bz2(self): + self.trinity("left1.fq.bz2", "fq") + self.assertEquals(30575, self.count_seqs(), "Unexpected sequence count") + + def test_fastq_multiple_files_single(self): + self.trinity("left1.fq,left1.fq.gz", "fq") + self.assertEquals(61150, self.count_seqs(), "Unexpected sequence count") + + def test_fastq_multiple_files_single_bz2(self): + self.trinity("left1.fq.bz2,left1.fq.gz", "fq") + self.assertEquals(61150, self.count_seqs(), "Unexpected sequence count") + + def test_fastq_multiple_files_single_reverse(self): + self.trinity("left1.fq,left1.fq.gz", "fq", True) + self.assertEquals(61150, self.count_seqs(), "Unexpected sequence count") + + def test_fasta(self): + self.trinity("left1.fa") + self.assertEquals(30575, self.count_seqs(), "Unexpected sequence count") + + def test_fasta_gz(self): + self.trinity("left1.fa.gz") + self.assertEquals(30575, self.count_seqs(), "Unexpected sequence count") + + def test_fasta_multiple_files_single(self): + self.trinity("left1.fa,left1.fa.gz") + self.assertEquals(61150, self.count_seqs(), "Unexpected sequence count") + + def test_fasta_multiple_files_single_reverse(self): + self.trinity("left1.fa,left1.fa.gz", reverse=True) + self.assertEquals(61150, self.count_seqs(), "Unexpected sequence count") + + def test_paired_fastq(self): + self.trinity("left1.fq", "fq", morefiles="right1.fq") + self.assertEquals(61150, self.count_seqs(), "Unexpected sequence count") + + def test_paired_fastq_gz(self): + self.trinity("left1.fq.gz", "fq", morefiles="right1.fq.gz") + self.assertEquals(61150, self.count_seqs(), "Unexpected sequence count") + + def test_fastq_multiple_files_paired(self): + self.trinity("left1.fq,left1.fq.gz", "fq", morefiles="right1.fq,right1.fq.gz") + self.assertEquals(122300, self.count_seqs(), "Unexpected sequence count") + + def test_fastq_multiple_files_paired_reverse(self): + self.trinity("left1.fq,left1.fq.gz", "fq", reverse=True, morefiles="right1.fq,right1.fq.gz") + self.assertEquals(122300, self.count_seqs(), "Unexpected sequence count") + + def test_fasta_paired(self): + self.trinity("left1.fa", morefiles="right1.fa") + self.assertEquals(61150, self.count_seqs(), "Unexpected sequence count") + + def test_paired_sequences_have_1_or_2_extension(self): + self.trinity("sra_test.fq", morefiles="sra_test2.fq", seqtype='fq') + self.assertEquals(0, self.count_bad_endings(), "Found sequences with bad endings") + + def test_fasta_gz_paired(self): + self.trinity("left1.fa.gz", morefiles="right1.fa.gz") + self.assertEquals(61150, self.count_seqs(), "Unexpected sequence count") + + def test_fasta_multiple_files_paired(self): + self.trinity("left1.fa,left1.fa.gz", morefiles="right1.fa,right1.fa.gz") + self.assertEquals(61150, self.count_seqs(), "Unexpected sequence count") + + def test_fasta_multiple_files_paired(self): + self.trinity("left1.fa,left1.fa.gz", morefiles="right1.fa,right1.fa.gz", reverse=True) + self.assertEquals(122300, self.count_seqs(), "Unexpected sequence count") + + def trinity(self, files, seqtype='fa', reverse=False, morefiles=None): + if morefiles: + tpl = "Trinity --left %s --right %s --prep --seqType %s --max_memory 2G --no_version_check --no_normalize_reads" + cmdline = tpl % (files, morefiles, seqtype) + else: + tpl = "Trinity --single %s --prep --seqType %s --max_memory 2G --no_version_check --no_normalize_reads" + cmdline = tpl % (files, seqtype) + if reverse: + cmdline += " --SS_lib_type " + ('RF' if morefiles else 'R') + print "Command line:", cmdline + with open("coverage.log", 'a') as file_out: + subprocess.call(cmdline,shell=True, stdout=file_out) + + def count_seqs(self): + f = "trinity_out_dir/single.fa" + if os.path.isfile(f): + handle = open(f, "rU") + else: + handle = open("trinity_out_dir/both.fa", "rU") + + seq_count = len([x for x in SeqIO.parse(handle, "fasta")]) + handle.close() + return seq_count + + def count_bad_endings(self): + f = "trinity_out_dir/single.fa" + if os.path.isfile(f): + handle = open(f, "rU") + else: + handle = open("trinity_out_dir/both.fa", "rU") + + seq_count = len(list(x for x in SeqIO.parse(handle, "fasta") if not (x.id.endswith('/1') or x.id.endswith('/2')))) + handle.close() + return seq_count + diff --git a/99.scripts/trinity_utils/util/support_scripts/tests/tests.py b/99.scripts/trinity_utils/util/support_scripts/tests/tests.py new file mode 100644 index 0000000..6b5a112 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/tests/tests.py @@ -0,0 +1,219 @@ +import subprocess +from Bio import SeqIO +import unittest +import shutil +import os +import time +import filecmp + +# Prereqs: +# module load bowtie/0.12.8 +# module load java +# module load samtools +# Trinity +# Copy the .gz files in sample_data/test_Trinity_Assembly to current directory +# Run using nosetests +MEM_FLAG = "--max_memory 2G" +TEMP_FILES = ['both.fa', 'inchworm.K25.L25.fa', 'jellyfish.kmers.fa'] + + +class TestTrinity(unittest.TestCase): + + @classmethod + def setUpClass(cls): + try: + os.remove('coverage.log') + except: + pass + + def tearDown(self): + shutil.rmtree('trinity_out_dir', True) + + def test_sample_data_seq_count(self): + print "When assembling the sample data, the number of sequences assembled should be between 75 and 100" + self.trinity( + "Trinity --seqType fq %s --left reads.left.fq.gz,reads2.left.fq.gz --right reads.right.fq.gz,reads2.right.fq.gz --SS_lib_type RF --CPU 4 --no_cleanup" % MEM_FLAG) + handle = open("trinity_out_dir/Trinity.fasta", "rU") + seq_count = len([x for x in SeqIO.parse(handle, "fasta")]) + handle.close() + self.assertTrue(85 <= seq_count <= 110, msg='Found %s sequences' % seq_count) + + def test_sample_data_trimmed_and_normalized(self): + print "When assembling the sample data with the --trimmomatic --normalize_reads flags, the number of sequences assembled should be between 75 and 85" + self.trinity( + "Trinity --seqType fq %s --left reads.left.fq.gz,reads2.left.fq.gz --right reads.right.fq.gz,reads2.right.fq.gz --SS_lib_type RF --CPU 4 --trimmomatic --normalize_reads --no_cleanup" % MEM_FLAG) + handle = open("trinity_out_dir/Trinity.fasta", "rU") + seq_count = len([x for x in SeqIO.parse(handle, "fasta")]) + handle.close() + self.assertTrue(85 <= seq_count <= 100, msg='Found %s sequences' % seq_count) + + def test_no_cleanup_leaves_temp_files(self): + print "The --no_cleanup flag should ensure that the output directory is left behind" + self.trinity( + "Trinity --seqType fq %s --left reads.left.fq.gz,reads2.left.fq.gz --right reads.right.fq.gz,reads2.right.fq.gz --SS_lib_type RF --CPU 4 --no_cleanup" % MEM_FLAG) + for f in TEMP_FILES: + self.assertTrue(os.path.exists("trinity_out_dir/%s" % f), msg="%s not found with no_cleanup" % f) + + def test_cleanup_removes_temp_files(self): + print "The --full_cleanup flag should ensure that the output directory is gone but the output file remains" + self.trinity( + "Trinity --seqType fq %s --left reads.left.fq.gz,reads2.left.fq.gz --right reads.right.fq.gz,reads2.right.fq.gz --SS_lib_type RF --CPU 4 --full_cleanup" % MEM_FLAG) + time.sleep(5) # Make sure the system has time to recognize the directory is gone + self.assertFalse(os.path.exists("trinity_out_dir"), msg="Did full_cleanup but trinity_out_dir exists") + self.assertTrue(os.path.isfile("trinity_out_dir.Trinity.fasta"), + msg="Did full_cleanup but output file not created") + + def test_single_end_with_rf_lib_type_error(self): + print "Single reads with an SS_lib_type of RF should result in an error" + try: + subprocess.call("Trinity --seqType fq --single reads.left.fq --SS_lib_type RF", shell=True) + except subprocess.CalledProcessError as e: + self.assertTrue("Error, with --single reads, the --SS_lib_type can be 'F' or 'R' only." in e.output) + + def test_single_end_with_fq(self): + print "Single reads with FQ file should succeed" + self.trinity("Trinity %s --seqType fq --single reads.left.fq --SS_lib_type F" % MEM_FLAG) + + def test_no_run_chrysalis(self): + print "The --no_run_chrysalis flag should result in no chrysalis subdirectory in the output directory" + self.trinity("Trinity %s --seqType fq --single reads.left.fq --SS_lib_type F --no_run_chrysalis" % MEM_FLAG) + self.assertEquals(0, len(os.listdir('trinity_out_dir/chrysalis'))) + + def test_no_run_inchworm(self): + print "The --no_run_inchworm flag should result in no inchworm.finished file" + self.trinity("Trinity %s --seqType fq --single reads.left.fq --SS_lib_type F --no_run_inchworm" % MEM_FLAG) + self.assertFalse(os.path.isfile("trinity_out_dir/inchworm.K25.L25.fa.finished"), + msg="Inchworm appears to have run although no_run_inchworm was specified") + self.assertTrue(os.path.isfile("trinity_out_dir/jellyfish.kmers.fa"), + msg="jellyfish.kmers.fa was not created") + + def test_no_bowtie(self): + print "The --no_bowtie flag should result in no bowtie.nameSorted.bam file" + self.trinity("Trinity %s --seqType fq --single reads.left.fq --SS_lib_type F --no_bowtie" % MEM_FLAG) + self.assertFalse(os.path.isfile("trinity_out_dir/bowtie.nameSorted.bam"), + msg="Bowtie appears to have run although no_bowtie was specified") + + def test_no_distributed_trinity_exec(self): + print "The --no_distributed_trinity_exec flag should run Jellyfish but not create an output file" + self.trinity("Trinity %s --seqType fq --single reads.left.fq --SS_lib_type F --no_distributed_trinity_exec" % MEM_FLAG) + self.assertTrue(os.path.isfile("trinity_out_dir/inchworm.K25.L25.fa.finished"), + msg="Inchworm did not appear to run with no_distributed_trinity_exec flag") + self.assertTrue(os.path.isfile("trinity_out_dir/jellyfish.kmers.fa.histo"), + msg="Jellyfish did not appear to run with no_distributed_trinity_exec flag") + self.assertFalse(os.path.isfile("trinity_out_dir/Trinity.fasta"), + msg="Trinity.fasta created with no_distributed_trinity_exec") + + def test_single_end_with_fa_and_reverse(self): + print "The --no_distributed_trinity_exec flag should run Jellyfish but not create an output file" + self.fq2fa() + self.trinity("Trinity %s --seqType fa --single reads.fa --SS_lib_type R" % MEM_FLAG) + + def test_output_correctly_changes_dir(self): + print "The --output flag should change the output directory" + shutil.rmtree('trinity_test', True) + self.trinity("Trinity %s --seqType fq --single reads.left.fq --SS_lib_type F --output trinity_test" % MEM_FLAG) + self.assertTrue(os.path.exists("trinity_test"), msg="Changed output directory but it was not created") + shutil.rmtree('trinity_test', True) + + def test_scaffold_iworm_contigs(self): + print "scaffold_iworm_contigs works as expected" + os.environ['LD_LIBRARY_PATH'] = os.environ['LD_LIBRARY_PATH'] + ':../src/trinity-plugins/htslib' + exe = "../src/trinity-plugins/scaffold_iworm_contigs/scaffold_iworm_contigs" + bamfile = "iworm.bowtie.nameSorted.bam" + ifile = "inchworm.K25.L25.fa" + f = subprocess.check_output([exe, bamfile, ifile]).split('\n') + expected_result = [['a340;25', '339', 'a9;40', '8', '41'], + ['a719;8', '718', 'a832;15', '831', '33'], + ['a1;43', '0', 'a346;23', '345', '31'], + ['a346;23', '345', 'a9;40', '8', '26'], + ['a339;142', '338', 'a37;14', '36', '25'], + ['a3;61', '2', 'a432;9', '431', '23'], + ['a345;34', '344', 'a40;12', '39', '21'], + ['a354;96', '353', 'a368;13', '367', '18'], + ['a689;4', '688', 'a774;6', '773', '13']] + + actual_result = [line.split('\t') for line in f if line] + self.assertEquals(expected_result, actual_result[0:9]) + actual_lengths = [int(s[4].strip()) for s in actual_result] + expected_order = list(reversed(sorted(actual_lengths))) + self.assertEquals(expected_order, actual_lengths) + + def test_Inchworm_handles_compressed_files(self): + print "A compressed single file should be handlred correctly by Inchworm" + self.trinity('Trinity %s --seqType fq --single reads.left.fq.gz --SS_lib_type F --no_run_chrysalis' % MEM_FLAG); + num_lines = sum(1 for line in open('trinity_out_dir/inchworm.K25.L25.fa')) + self.assertTrue(2850 <= num_lines <= 3100, msg='Found %s lines' % num_lines) + + def test_QuantifyGraph_works(self): + exe = "../src/Chrysalis/QuantifyGraph" + args = " -g Cbin0/c0.graph.tmp -i Cbin0/c0.reads.tmp -o c0.graph.out -max_reads 200000 -k 24" + result = subprocess.call(exe + args,shell=True) + self.assertEquals(0, result, "QuantifyGraph failed") + self.assertTrue(filecmp.cmp("c0.graph.out", "Cbin0/c0.graph.out.master", shallow=False), "Unexpected result from QuantifyGraph") + +### information tests + def test_cite(self): + print "Cite flag should return citing information" + expected = '\n\n* Trinity:\nFull-length transcriptome assembly from RNA-Seq data without a reference genome.\nGrabherr MG, Haas BJ, Yassour M, Levin JZ, Thompson DA, Amit I, Adiconis X, Fan L,\nRaychowdhury R, Zeng Q, Chen Z, Mauceli E, Hacohen N, Gnirke A, Rhind N, di Palma F,\nBirren BW, Nusbaum C, Lindblad-Toh K, Friedman N, Regev A.\nNature Biotechnology 29, 644\xe2\x80\x93652 (2011)\nPaper: http://www.nature.com/nbt/journal/v29/n7/full/nbt.1883.html\nCode: http://trinityrnaseq.sf.net\n\n\n' + cite = subprocess.check_output(["Trinity", "--cite"]) + self.assertEqual(expected, cite) + + def test_version(self): + print "Version flag should return version information" + try: + subprocess.check_output(["Trinity", "--version"]) + self.fail("Version returned 0 errorcode!") + except subprocess.CalledProcessError as e: + self.assertTrue('Trinity version: __TRINITY_VERSION_TAG__' in e.output) + self.assertTrue('using Trinity devel version. Note, latest production release is: v2.1.1' in e.output) + + + def test_show_full_usage_info(self): + print "show_full_usage_info flag has several option sections" + try: + subprocess.check_output(["Trinity", "--show_full_usage_info"]) + except subprocess.CalledProcessError as e: + self.assertTrue("Inchworm and K-mer counting-related options" in e.output) + self.assertTrue("Chrysalis-related options" in e.output) + self.assertTrue("Butterfly-related options" in e.output) + self.assertTrue("Quality Trimming Options" in e.output) + self.assertTrue("In silico Read Normalization Options" in e.output) + +### Invalid command line tests + def test_no_JM_specified_error(self): + print "max_memory flag is required" + error = self.get_error("Trinity --seqType fq --single reads.left.fq --SS_lib_type F") + self.assertTrue("Error, must specify max memory for jellyfish to use, eg. --max_memory 10G" in error) + + def test_invalid_option_error(self): + print "Invalid options result in an error" + error = self.get_error("Trinity --squidward") + self.assertTrue("ERROR, don't recognize parameter: --squidward" in error) + + def test_set_no_cleanup_and_full_cleanup_error(self): + print "Setting no_cleanup and full_cleanup together results in an error" + error = self.get_error("Trinity --no_cleanup --full_cleanup") + self.assertTrue("cannot set --no_cleanup and --full_cleanup as they contradict" in error) + + +### Helper methods + def trinity(self, cmdline): + print "Command line:", cmdline + with open("coverage.log", 'a') as file_out: + file_out.write("COMMAND: %s\n" % cmdline) + file_out.flush() + subprocess.call(cmdline,shell=True, stdout=file_out) + file_out.write("TEST COMPLETE\n") + + def get_error(self, cmd): + try: + subprocess.check_output(cmd.split(' ')) + except subprocess.CalledProcessError as e: + return e.output + + def fq2fa(self): + handle = open("reads.left.fq", "rU") + records = [x for x in SeqIO.parse(handle, "fastq")] + handle.close() + SeqIO.write(records, "reads.fa", "fasta") + diff --git a/99.scripts/trinity_utils/util/support_scripts/trinity_install_tests.sh b/99.scripts/trinity_utils/util/support_scripts/trinity_install_tests.sh new file mode 100644 index 0000000..d1a2342 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/trinity_install_tests.sh @@ -0,0 +1,50 @@ +#!/bin/bash + +echo "~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~" +echo "" +echo 'Performing Unit Tests of Build' +echo ' ' +echo "~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~" + + +if [ -e "Inchworm/bin/inchworm" ] +then + echo "Inchworm: has been Installed Properly" +else + echo "Inchworm Installation appears to have FAILED" +fi +if [ -e "Chrysalis/bin/Chrysalis" ] +then + echo "Chrysalis: has been Installed Properly" +else + echo "Chrysalis Installation appears to have FAILED" +fi +if [ -e "Chrysalis/bin/QuantifyGraph" ] +then + echo "QuantifyGraph: has been Installed Properly" +else + echo "QuantifyGraph Installation appears to have FAILED" +fi + +if [ -e "Chrysalis/bin/GraphFromFasta" ] +then + echo "GraphFromFasta: has been Installed Properly" +else + echo "GraphFromFasta Installation appears to have FAILED" +fi + +if [ -e "Chrysalis/bin/ReadsToTranscripts" ] +then + echo "ReadsToTranscripts: has been Installed Properly" +else + echo "ReadsToTranscripts Installation appears to have FAILED" +fi + + +if [ -e "trinity-plugins/BIN/ParaFly" ] +then + echo "parafly: has been Installed Properly" +else + echo "parafly Installation appears to have FAILED" +fi + diff --git a/99.scripts/trinity_utils/util/support_scripts/trinity_installer.py b/99.scripts/trinity_utils/util/support_scripts/trinity_installer.py new file mode 100644 index 0000000..1d82e63 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/trinity_installer.py @@ -0,0 +1,25 @@ +#!/usr/bin/env python + +import os, re, sys, subprocess + +trinity_package_dir = os.path.abspath(sys.argv[0] + "/../../../") +print("Trinity package dir: {}".format(trinity_package_dir)) + +trinity_package_name = os.path.basename(trinity_package_dir) +print("Trinity package name: {}".format(trinity_package_name)) + +destination_package_dir = "/usr/local/bin" + +subprocess.check_call("rsync -av --exclude='.*' {}/ {}".format(trinity_package_dir, destination_package_dir), shell=True) + +print("Trinity package installed at: {}".format(destination_package_dir)) +print("\n\n\tFor convenience, set env var TRINITY_HOME={}".format(destination_package_dir)) +print("\n\tSimply add:\n\n\texport TRINITY_HOME={}".format(destination_package_dir)) +print("\tto your ~/.bashrc file.\n\n") +print("\tand run trinity via: $TRINITY_HOME/Trinity\n\n\n") + +sys.exit(0) + + + + diff --git a/99.scripts/trinity_utils/util/support_scripts/wig_clip_to_bed.pl b/99.scripts/trinity_utils/util/support_scripts/wig_clip_to_bed.pl new file mode 100644 index 0000000..d31f67c --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/wig_clip_to_bed.pl @@ -0,0 +1,70 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use lib ($ENV{EUK_MODULES}); +use WigParser; +use Gene_obj; + + +my $usage = "usage: $0 clip.wig\n\n"; + +my $clip_wig = $ARGV[0] or die $usage; + +main: { + + + my %mol_to_clips = &get_clip_pts($clip_wig); + + foreach my $mol (keys %mol_to_clips) { + + my @clip_pos = @{$mol_to_clips{$mol}}; + + foreach my $pos (@clip_pos) { + + my ($end5, $end3) = ($pos-1, $pos+1); + + my $gene_obj = new Gene_obj(); + $gene_obj->populate_gene_object( {$end5=>$end3}, {$end5=>$end3}); + + $gene_obj->{asmbl_id} = $mol; + + $gene_obj->{com_name} = "$mol-C:$pos"; + + print $gene_obj->to_BED_format(); + } + } + + exit(0); +} + + +#### +sub get_clip_pts { + my ($jaccard_clips) = @_; + + my %trans_to_clips; + + my $trans_acc = ""; + + open (my $fh, $jaccard_clips) or die "Error, cannot open file $jaccard_clips"; + while (<$fh>) { + chomp; + if (/^variableStep chrom=(\S+)/) { + $trans_acc = $1; + } + elsif (/^(\d+)/) { + my ($coord, $val) = split(/\t/); + if ($val) { + push (@{$trans_to_clips{$trans_acc}}, $coord); + } + } + } + close $fh; + + return(%trans_to_clips); +} + + + diff --git a/99.scripts/trinity_utils/util/support_scripts/write_partitioned_trinity_cmds.pl b/99.scripts/trinity_utils/util/support_scripts/write_partitioned_trinity_cmds.pl new file mode 100644 index 0000000..aa5c196 --- /dev/null +++ b/99.scripts/trinity_utils/util/support_scripts/write_partitioned_trinity_cmds.pl @@ -0,0 +1,100 @@ +#!/usr/bin/env perl + +use strict; +use warnings; + +use FindBin; +use Getopt::Long qw(:config no_ignore_case bundling pass_through); + +my $usage = <<__EOUSAGE__; + +#################################################################################### +# +# usage: $0 --reads_list_file [Trinity params] +# +# Required: +# +# --reads_list_file file containing list of filenames corresponding +# to the reads.fasta +# +# Optional for singularity usage: +# +# --singularity_img path to singularity image +# +# --singularity_extra_params singularity extra parameters to include +# +##################################################################################### + + +__EOUSAGE__ + + ; + + +my $reads_file; +my $help_flag; +my $singularity_img; +my $singularity_extra_params = ""; + + +&GetOptions ( + 'reads_list_file=s' => \$reads_file, + 'h' => \$help_flag, + 'singularity_img=s' => \$singularity_img, + 'singularity_extra_params=s' => \$singularity_extra_params, + ); + + +my @TRIN_ARGS = @ARGV; + +if ($help_flag) { + die $usage; +} + +unless ($reads_file && -e $reads_file) { + die $usage; +} + +unless (-s $reads_file) { + die "Error, reads file listing: $reads_file is empty. This tends to happen when there were too few reads to assemble. "; +} + + +my $trin_args = ""; +while (@TRIN_ARGS) { + my $arg = shift @TRIN_ARGS; + + if ($arg =~ /bfly_opts/) { + my $val = shift @TRIN_ARGS; + # retain quotes around multiparams + + $trin_args .= "$arg \"$val\" "; + } + else { + $trin_args .= "$arg "; + } +} + + +open (my $fh, $reads_file) or die "Error, cannot open file $reads_file"; +while (<$fh>) { + chomp; + my @x = split(/\s+/); + + my $file = pop @x; + + + my $cmd = "$FindBin::RealBin/../../Trinity --single \"$file\" --output \"$file.out\" $trin_args "; + + if ($singularity_img) { + $cmd = "singularity exec $singularity_extra_params $singularity_img $cmd"; + } + print "$cmd\n"; +} + +exit(0); + + + + + diff --git a/99.scripts/workflow/orthology_inference/01.orthofinder_sc_align.sh b/99.scripts/workflow/orthology_inference/01.orthofinder_sc_align.sh new file mode 100644 index 0000000..a6b069c --- /dev/null +++ b/99.scripts/workflow/orthology_inference/01.orthofinder_sc_align.sh @@ -0,0 +1,8 @@ +#! /usr/bin/env bash +mkdir -p msa +echo -n > mafft.cmds +for i in ogs/*.fa ; do + j=$(basename "$i") + echo "linsi --quiet $i > msa/$j" >> mafft.cmds +done +xargs -t -P 8 -I cmd -a mafft.cmds bash -c "cmd" diff --git a/99.scripts/workflow/orthology_inference/02.hmmbuild.sh b/99.scripts/workflow/orthology_inference/02.hmmbuild.sh new file mode 100644 index 0000000..e5b1f4e --- /dev/null +++ b/99.scripts/workflow/orthology_inference/02.hmmbuild.sh @@ -0,0 +1,8 @@ +#! /usr/bin/env bash +mkdir -p hmms +echo -n > hmmbuild.cmds +for i in msa/*.fa ; do + j=$(basename "$i") + echo "hmmbuild -o hmms/${j}.hmmbuild.out --amino hmms/${j}.hmm $i" >> hmmbuild.cmds +done +xargs -t -P 8 -I cmd -a hmmbuild.cmds bash -c "cmd" \ No newline at end of file diff --git a/99.scripts/workflow/orthology_inference/03.hmmsearch.sh b/99.scripts/workflow/orthology_inference/03.hmmsearch.sh new file mode 100644 index 0000000..88996be --- /dev/null +++ b/99.scripts/workflow/orthology_inference/03.hmmsearch.sh @@ -0,0 +1,8 @@ +#! /usr/bin/env bash +mkdir -p hmmsearch +echo -n > hmmsearch.cmds +for i in hmms/*.hmm ; do + j=$(basename "$i") + echo "hmmsearch --tblout hmmsearch/${j}search.tblout $i ../../01.reference/Zju.pep.fa > hmmsearch/${j}search.rawout" >> hmmsearch.cmds +done +xargs -t -P 8 -I cmd -a hmmsearch.cmds bash -c "cmd" diff --git a/99.scripts/workflow/orthology_inference/04.pep_align.sh b/99.scripts/workflow/orthology_inference/04.pep_align.sh new file mode 100755 index 0000000..da82fab --- /dev/null +++ b/99.scripts/workflow/orthology_inference/04.pep_align.sh @@ -0,0 +1,8 @@ +#! /usr/bin/env bash +mkdir -p pep_aln +echo -n > mafft.cmds +for i in raw_ogs/pep/*.fa; do + j=$(basename "$i") + echo "linsi --quiet $i > pep_aln/${j/.fa/.pal}" >> mafft.cmds +done +xargs -t -P 8 -I cmd -a mafft.cmds bash -c "cmd" diff --git a/99.scripts/workflow/orthology_inference/05.pal2nal.sh b/99.scripts/workflow/orthology_inference/05.pal2nal.sh new file mode 100755 index 0000000..3703ea8 --- /dev/null +++ b/99.scripts/workflow/orthology_inference/05.pal2nal.sh @@ -0,0 +1,8 @@ +#! /usr/bin/env bash +mkdir -p cds_aln +echo -n > pal2nal.cmds +for i in pep_aln/*.pal; do + j=$(basename "$i") + echo "pal2nal.pl $i raw_ogs/cds/${j/.pal/.fa} -output fasta > cds_aln/${j/.pal/.nal}" >> pal2nal.cmds +done +xargs -t -P 8 -I cmd -a pal2nal.cmds bash -c "cmd" diff --git a/99.scripts/workflow/orthology_inference/06.trim_alignment.sh b/99.scripts/workflow/orthology_inference/06.trim_alignment.sh new file mode 100755 index 0000000..80431ba --- /dev/null +++ b/99.scripts/workflow/orthology_inference/06.trim_alignment.sh @@ -0,0 +1,8 @@ +#! /usr/bin/env bash +mkdir -p trimed_nal +echo -n > trimal.cmds +for i in cds_aln/*.nal ;do + j=$(basename "$i") + echo "trimal -in $i -out trimed_nal/${j/.nal/.trimed.fa} -automated1 -resoverlap 0.5 -seqoverlap 50" >> trimal.cmds +done +xargs -t -P 4 -I cmd -a trimal.cmds bash -c "cmd" diff --git a/99.scripts/workflow/orthology_inference/07.fasttree.sh b/99.scripts/workflow/orthology_inference/07.fasttree.sh new file mode 100755 index 0000000..66edf59 --- /dev/null +++ b/99.scripts/workflow/orthology_inference/07.fasttree.sh @@ -0,0 +1,8 @@ +#! /usr/bin/env bash +mkdir -p fasttree +echo -n > fasttree.cmds +for i in trimed_nal/*.trimed.fa ;do + j=$(basename "$i") + echo "FastTree -nt -gtr -quiet $i > fasttree/${j/.trimed.fa/.tree}" >> fasttree.cmds +done +xargs -t -P 8 -I cmd -a fasttree.cmds bash -c "cmd" diff --git a/99.scripts/workflow/orthology_inference/08.treeshrink.sh b/99.scripts/workflow/orthology_inference/08.treeshrink.sh new file mode 100755 index 0000000..b11bfb7 --- /dev/null +++ b/99.scripts/workflow/orthology_inference/08.treeshrink.sh @@ -0,0 +1,11 @@ +#! /usr/bin/env bash +mkdir -p treeshrink +for i in trimed_nal/*.trimed.fa; do + j=$(basename "$i") + mkdir -p treeshrink/"${j/.trimed.fa/}" + cd treeshrink/"${j/.trimed.fa/}" || exit 1 + ln -s ../../fasttree/"${j/.trimed.fa/.tree}" input.tree + ln -s ../../"$i" input.fasta + cd ../../ +done +run_treeshrink.py -i treeshrink/ -t input.tree -a input.fasta > treeshrink.log diff --git a/99.scripts/workflow/orthology_inference/09.filter_ogs.sh b/99.scripts/workflow/orthology_inference/09.filter_ogs.sh new file mode 100755 index 0000000..ac14150 --- /dev/null +++ b/99.scripts/workflow/orthology_inference/09.filter_ogs.sh @@ -0,0 +1,12 @@ +#! /usr/bin/env bash +total_taxon=11 +min_seq_length=300 +mkdir -p final_ogs +for i in treeshrink/* ; do + j=$(basename "$i") + seqlen=$(seqkit fx2tab -C ATCG "$i"/output.fasta | awk '{print $3}' | sort -n | head -n 1) + seqnum=$(grep -c ">" "$i"/output.fasta) + if [[ $seqnum -eq $total_taxon && $seqlen -ge $min_seq_length ]]; then + cp -l "$i"/output.fasta final_ogs/"${j}.fa" + fi +done diff --git a/99.scripts/workflow/phylogeny_reconstruction/01.modeltest.sh b/99.scripts/workflow/phylogeny_reconstruction/01.modeltest.sh new file mode 100755 index 0000000..3ef0126 --- /dev/null +++ b/99.scripts/workflow/phylogeny_reconstruction/01.modeltest.sh @@ -0,0 +1,7 @@ +#! /usr/bin/env bash +mkdir -p modeltests +for i in ../gene_alignment/*.fa ; do + j=$(basename "$i") + echo "modeltest-ng -p 2 -r 12345 --force -i $i -d nt -t ml -o modeltests/${j/.fa/}.modeltest" >> modeltest.cmds +done +xargs -t -P 4 -I cmd -a modeltest.cmds bash -c "cmd" diff --git a/99.scripts/workflow/phylogeny_reconstruction/02.raxml_per_gene.sh b/99.scripts/workflow/phylogeny_reconstruction/02.raxml_per_gene.sh new file mode 100755 index 0000000..9736f0f --- /dev/null +++ b/99.scripts/workflow/phylogeny_reconstruction/02.raxml_per_gene.sh @@ -0,0 +1,9 @@ +#! /usr/bin/env bash +mkdir -p raxml_ng +echo -n > raxml_ng.cmds +for i in modeltests/*.modeltest.out ; do + j=$(basename "$i") + cmd=$(grep "raxml-ng" "$i" | tail -n 1 | sed 's/> //') + echo "$cmd --all --bs-trees 1000 --outgroup Zju --redo --threads 4 --seed 12345 --prefix raxml_ng/${j/.modeltest.out/} > /dev/null" >> raxml_ng.cmds +done +xargs -t -P 3 -I cmd -a raxml_ng.cmds bash -c "cmd" diff --git a/99.scripts/workflow/phylogeny_reconstruction/03.snaq.jl b/99.scripts/workflow/phylogeny_reconstruction/03.snaq.jl new file mode 100644 index 0000000..9376967 --- /dev/null +++ b/99.scripts/workflow/phylogeny_reconstruction/03.snaq.jl @@ -0,0 +1,89 @@ +#! /usr/bin/env julia +## Installing Dependencies +## using Pkg +## Pkg.add("Distributed") +## Pkg.add("DataFrames") +## Pkg.add("CSV") +## Pkg.add("SNaQ") +## Pkg.add("PhyloNetworks") +## Pkg.add("RCall") +## Pkg.add("PhyloPlots") +## Pkg.add("QuartetNetworkGoodnessFit") + +# Running SNaQ Analysis +using PhyloNetworks, SNaQ; +using Distributed; +addprocs(5); +@everywhere using PhyloNetworks, SNaQ; +nruns = 100; # number of runs for each hmax +astralfile = joinpath("..", "..", "species_tree", "aster.out"); +astraltree = readnewick(astralfile); + +### Reading RAxML gene trees and ASTRAL species tree +### running in raxml_snaq/ folder +# raxmltrees = joinpath("..", "..", "species_tree", "all.trees"); +# inputCF = readtrees2CF(raxmltrees); +# net0 = snaq!(astraltree, inputCF, hmax=0, filename="net0", seed=123, outgroup="Zju", runs=nruns); +# net1 = snaq!(net0, inputCF, hmax=1, filename="net1", seed=123, outgroup="Zju", runs=nruns); +# net2 = snaq!(net1, inputCF, hmax=2, filename="net2", seed=123, outgroup="Zju", runs=nruns); +# net3 = snaq!(net2, inputCF, hmax=3, filename="net3", seed=123, outgroup="Zju", runs=nruns); +# net4 = snaq!(net3, inputCF, hmax=4, filename="net4", seed=123, outgroup="Zju", runs=nruns); + +### Alternatively, reading in the input files from Bucky +### running in input_snaq/ folder +inputCFfile = joinpath("..", "..", "input", "input.CFs.csv"); +inputCF = readtableCF(inputCFfile); +net0 = snaq!(astraltree, inputCF, hmax=0, filename="net0", seed=123, outgroup="Zju", runs=nruns); +net1 = snaq!(net0, inputCF, hmax=1, filename="net1", seed=123, outgroup="Zju", runs=nruns); +net2 = snaq!(net1, inputCF, hmax=2, filename="net2", seed=123, outgroup="Zju", runs=nruns); +net3 = snaq!(net2, inputCF, hmax=3, filename="net3", seed=123, outgroup="Zju", runs=nruns); +net4 = snaq!(net3, inputCF, hmax=4, filename="net4", seed=123, outgroup="Zju", runs=nruns); + + +# Plotting the SNaQ results +using PhyloPlots, RCall; +## Network scores vs. hmax +scores = [loglik(net0), loglik(net1), loglik(net2), loglik(net3), loglik(net4)]; +hmax = collect(0:4); +R"pdf"("snaq_network_scores.pdf", width=12, height=8); +R"plot"(hmax, scores, type="b", ylab="network score", xlab="hmax", col="blue"); +R"dev.off"(); + +## Rerooting and rotating the networks for better visualization +rootatnode!(net1, "Zju"); +rootatnode!(net2, "Zju"); +rootatnode!(net3, "Zju"); +rootatnode!(net4, "Zju"); +### rotate!(net1, -2); +### rotate!(net2, -2); +### rotate!(net3, -2); +### rotate!(net4, -2); + +## Plotting the networks +R"pdf"("snaq_networks.pdf", width=14, height=10); +R"layout(matrix(1:4, 2, 2, byrow=TRUE))"; # to get 4 plots into a single figure: 2 row, 2 columns +R"par"(mar=[0, 0, 1.5, 0]); # for smaller margins +xmin, xmax = PhyloPlots.PhyloPlots.edgenode_coordinates(net1, false, false)[13:14]; +xmax += (xmax - xmin) * 0.3; +plot(net1, showgamma=true, tipoffset=0.1, xlim=[xmin, xmax]); +R"mtext"(string("hmax=1, loglik=-", round(loglik(net1), digits=2)), font=2); +xmin, xmax = PhyloPlots.PhyloPlots.edgenode_coordinates(net2, false, false)[13:14]; +xmax += (xmax - xmin) * 0.3; +plot(net2, showgamma=true, tipoffset=0.1, xlim=[xmin, xmax]); +R"mtext"(string("hmax=2, loglik=-", round(loglik(net2), digits=2)), font=2); +xmin, xmax = PhyloPlots.PhyloPlots.edgenode_coordinates(net3, false, false)[13:14]; +xmax += (xmax - xmin) * 0.3; +plot(net3, showgamma=true, tipoffset=0.1, xlim=[xmin, xmax]); +R"mtext"(string("hmax=3, loglik=-", round(loglik(net3), digits=2)), font=2); +xmin, xmax = PhyloPlots.PhyloPlots.edgenode_coordinates(net4, false, false)[13:14]; +xmax += (xmax - xmin) * 0.3; +plot(net4, showgamma=true, tipoffset=0.1, xlim=[xmin, xmax]); +R"mtext"(string("hmax=4, loglik=-", round(loglik(net4), digits=2)), font=2); +R"dev.off"(); + +## expected vs. observed quartet concordance factors +using CSV, DataFrames; + +# Goodness of fit of the SNaQ networks +using QuartetNetworkGoodnessFit; + diff --git a/99.scripts/workflow/phylogeny_reconstruction/04.concatenate_ogs.py b/99.scripts/workflow/phylogeny_reconstruction/04.concatenate_ogs.py new file mode 100755 index 0000000..0487d6b --- /dev/null +++ b/99.scripts/workflow/phylogeny_reconstruction/04.concatenate_ogs.py @@ -0,0 +1,146 @@ +#! /usr/bin/env python3 +import os +from Bio import SeqIO +from Bio.Seq import Seq +from Bio.SeqRecord import SeqRecord +from collections import defaultdict +import argparse + + +def get_sequence_lengths(fasta_files): + """ + get the lengths of sequences in each FASTA file + Assumes all sequences in a file have the same length + """ + file_lengths = {} + for fasta_file in fasta_files: + try: + with open(fasta_file, "r") as f: + for record in SeqIO.parse(f, "fasta"): + # get length of the first sequence + file_lengths[fasta_file] = len(record.seq) + break + except Exception as e: + print(f"Error reading file {fasta_file}: {e}") + file_lengths[fasta_file] = 0 + + return file_lengths + + +def concatenate_fasta_files(fasta_files, output_file): + """ + Concatenate sequences from multiple FASTA files by name, using "-" for missing sequences. + """ + # Get the sequence lengths for each file + file_lengths = get_sequence_lengths(fasta_files) + + # Store all sequence names and their corresponding content + sequences_dict = defaultdict(dict) + all_sequence_names = set() + + # Read sequences from each file + for i, fasta_file in enumerate(fasta_files): + try: + with open(fasta_file, "r") as f: + for record in SeqIO.parse(f, "fasta"): + seq_name = record.id + sequences_dict[seq_name][i] = str(record.seq) + all_sequence_names.add(seq_name) + except Exception as e: + print(f"Error reading file {fasta_file}: {e}") + + # Create concatenated sequences + concatenated_sequences = [] + + for seq_name in sorted(all_sequence_names): + concatenated_seq = [] + + for i, fasta_file in enumerate(fasta_files): + if i in sequences_dict[seq_name]: + # This file has the sequence, add it directly + concatenated_seq.append(sequences_dict[seq_name][i]) + else: + # This file is missing the sequence, use "-" to fill the gap + gap_length = file_lengths[fasta_file] + concatenated_seq.append("-" * gap_length) + + # Concatenate all parts of the sequence + full_sequence = "".join(concatenated_seq) + + # Create a new sequence record + + new_record = SeqRecord( + Seq(full_sequence), + id=seq_name, + description=f"concatenated_from_{len(fasta_files)}_files", + ) + concatenated_sequences.append(new_record) + + with open(output_file, "w") as output_handle: + SeqIO.write(concatenated_sequences, output_handle, "fasta") + + print( + f"Successfully concatenate {len(concatenated_sequences)} sequences to {output_file}" + ) + print(f"Input file count: {len(fasta_files)}") + + # Output statistics + for i, fasta_file in enumerate(fasta_files): + seq_count = sum(1 for seqs in sequences_dict.values() if i in seqs) + print( + f"File {i + 1}: {os.path.basename(fasta_file)} - Sequence count: {seq_count}, Sequence length: {file_lengths[fasta_file]}." + ) + + print(f"Total output sequence length: {len(concatenated_sequences[0].seq)}.") + + +def get_fasta_files_from_directory(directory, extensions): + """ + get all FASTA files from a directory with specified extensions + """ + fasta_files = [] + for filename in os.listdir(directory): + if any(filename.endswith(ext) for ext in extensions): + fasta_files.append(os.path.join(directory, filename)) + return sorted(fasta_files) + + +def main(): + parser = argparse.ArgumentParser( + description="Concatenate multiple FASTA files by sequence names." + ) + parser.add_argument("-i", "--input", nargs="+", help="Input FASTA file list") + parser.add_argument("-d", "--directory", help="Directory containing FASTA files") + parser.add_argument("-o", "--output", required=True, help="Output file") + parser.add_argument( + "-e", + "--extensions", + nargs="+", + default=[".fasta", ".fa", ".fna"], + help="FASTA file extensions to look for in directory", + ) + + args = parser.parse_args() + + # 获取输入文件 + if args.directory: + fasta_files = get_fasta_files_from_directory(args.directory, args.extensions) + if not fasta_files: + print( + f"Cannot find FASTA files in {args.directory} with extensions {args.extensions}" + ) + return + elif args.input: + fasta_files = args.input + else: + print("Please specify input files or directory") + return + + print(f"Found {len(fasta_files)} FASTA files:") + + # Perform concatenation + concatenate_fasta_files(fasta_files, args.output) + + +if __name__ == "__main__": + main() diff --git a/99.scripts/workflow/phylogeny_reconstruction/05.densitree.r b/99.scripts/workflow/phylogeny_reconstruction/05.densitree.r new file mode 100755 index 0000000..048ca40 --- /dev/null +++ b/99.scripts/workflow/phylogeny_reconstruction/05.densitree.r @@ -0,0 +1,26 @@ +#! /usr/bin/env Rscript +# DensiTree visualization of phylogenetic trees +args <- commandArgs(trailingOnly = TRUE) +if (length(args) != 6) { + stop("Usage: Rscript 05.densitree.r ") +} +tree_file <- args[1] +tip_order_file <- args[2] +root <- args[3] +output_pdf <- args[4] +width <- as.numeric(args[5]) +height <- as.numeric(args[6]) + +library(ape) +library(phangorn) +trees <- read.tree(tree_file) +pdf(output_pdf, width = width, height = height) +for(i in 1:length(trees)) { + trees[[i]] <- compute.brlen(root(trees[[i]], root)) +} +tip_order <- rev(readLines(tip_order_file)) +densiTree(trees, consensus=tip_order, alpha = 0.01, + col = "#009900", type = "cladogram", + label.offset = 0.02, scale.bar = FALSE +) +dev.off() diff --git a/99.scripts/workflow/phylogeny_reconstruction/06.alignments_to_nexus.sh b/99.scripts/workflow/phylogeny_reconstruction/06.alignments_to_nexus.sh new file mode 100755 index 0000000..2dee5cd --- /dev/null +++ b/99.scripts/workflow/phylogeny_reconstruction/06.alignments_to_nexus.sh @@ -0,0 +1,18 @@ +#! /usr/bin/env bash + +if [ "$#" -ne 3 ]; then + echo "Usage: $0 " + exit 1 +fi + +input_dir=$1 +extension=$2 +output_dir=$3 +mkdir -p "${output_dir}" +for f in "${input_dir}"/*."${extension}"; do + filename=$(basename -- "${f}") + filename_noext="${filename%.*}" + output_file="${output_dir}/${filename_noext}.nex" + echo "Converting ${f} to ${output_file}" + seqmagick convert --output-format nexus --alphabet dna --input-format fasta "${f}" "${output_file}" +done diff --git a/99.scripts/workflow/phylogeny_reconstruction/07.mbsum.sh b/99.scripts/workflow/phylogeny_reconstruction/07.mbsum.sh new file mode 100755 index 0000000..7d765fb --- /dev/null +++ b/99.scripts/workflow/phylogeny_reconstruction/07.mbsum.sh @@ -0,0 +1,21 @@ +#!/usr/bin/env bash + +mkdir -p ../mbsum_out +echo -n "" > ../mbsum.log +for i in *.nex.tar.gz; do + base=$(basename "$i" .nex.tar.gz) + echo "Processing ${base}" >> ../mbsum.log + mkdir -p "${base}" + tar -xzf "$i" -C "${base}" + ## skip Average standard deviation of split frequencies > 0.01 + dsf=$(awk '/Average standard deviation of split frequencies:/ {out=$7} END{print out+0}' "${base}"/*.nex.log 2>/dev/null) + if awk -v d="$dsf" 'BEGIN{if (d >= 0.01) exit 0; exit 1}'; then + echo "Skipping ${base} due to high DSF: ${dsf}" >> ../mbsum.log + continue + else + mbsum "${base}"/*.t -n 1000 -o ../mbsum_out/"${base}".in >> ../mbsum.log 2>&1 + fi + rm -rf "${base}" + echo "Completed ${base}" >> ../mbsum.log +done +echo "All done!" diff --git a/99.scripts/workflow/phylogeny_reconstruction/mbblock.txt b/99.scripts/workflow/phylogeny_reconstruction/mbblock.txt new file mode 100644 index 0000000..e32a24d --- /dev/null +++ b/99.scripts/workflow/phylogeny_reconstruction/mbblock.txt @@ -0,0 +1,12 @@ +begin mrbayes; +set nowarnings=yes; +set usebeagle=yes; +set autoclose=yes; +set seed=12345; +set swapseed=12345; +lset nst=6 rates=gamma; +mcmcp ngen=2000000 burninfrac=.25 samplefreq=1000 printfreq=10000 checkpoint=no +diagnfreq=10000 nruns=3 nchains=3 temp=0.40 swapfreq=10 stoprule=no; +mcmc; +sumt; +end; \ No newline at end of file diff --git a/pixi.lock b/pixi.lock new file mode 100644 index 0000000..f01a2e0 --- /dev/null +++ b/pixi.lock @@ -0,0 +1,9689 @@ +version: 6 +environments: + default: + channels: + - url: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/ + - url: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/ + - url: https://conda.anaconda.org/conda-forge/ + - url: https://conda.anaconda.org/bioconda/ + packages: + linux-64: + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/_libgcc_mutex-0.1-conda_forge.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/_openmp_mutex-4.5-2_gnu.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/_r-mutex-1.0.1-anacondar_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/alsa-lib-1.2.14-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/archspec-0.2.5-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/argcomplete-3.6.3-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/aria2-1.37.0-hbc8128a_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/arpack-3.9.1-nompi_hf03ea27_102.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/aster-1.23-h9948957_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/attr-2.5.2-h39aace5_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/augustus-3.5.0-pl5321h57ba348_8.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bamtools-2.5.3-he132191_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bbmap-39.37-he5f24ec_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/beagle-lib-4.0.1-h9948957_3.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/beast-10.5.0-hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/binutils_impl_linux-64-2.45-h9d8b0ac_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-annotate-1.84.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-annotationdbi-1.68.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-biobase-2.66.0-r44h3df3fcb_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-biocfilecache-2.14.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-biocgenerics-0.52.0-r44hdfd78af_3.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-biocio-1.16.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-biocparallel-1.40.0-r44he5774e6_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-biomart-2.62.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-biostrings-2.74.0-r44h3df3fcb_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-ctc-1.80.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-data-packages-20250625-hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-delayedarray-0.32.0-r44h3df3fcb_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-deseq2-1.46.0-r44he5774e6_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-dexseq-1.52.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-edger-4.4.0-r44h3df3fcb_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-genefilter-1.88.0-r44h81e381d_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-genelendatabase-1.42.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-geneplotter-1.84.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-genomeinfodb-1.42.0-r44hdfd78af_2.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-genomeinfodbdata-1.2.13-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-genomicalignments-1.42.0-r44h3df3fcb_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-genomicfeatures-1.58.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-genomicranges-1.58.0-r44h3df3fcb_2.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-go.db-3.20.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-goseq-1.58.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-iranges-2.40.0-r44h3df3fcb_2.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-keggrest-1.46.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-limma-3.62.1-r44h15a9599_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-matrixgenerics-1.18.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-qvalue-2.38.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-rhtslib-3.2.0-r44h15a9599_2.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-rsamtools-2.22.0-r44h77050f0_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-rtracklayer-1.66.0-r44h15a9599_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-s4arrays-1.6.0-r44h3df3fcb_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-s4vectors-0.44.0-r44h3df3fcb_2.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-seqlogo-1.72.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-sparsearray-1.6.0-r44h3df3fcb_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-summarizedexperiment-1.36.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-txdbmaker-1.2.0-r44hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-ucsc.utils-1.2.0-r44h9ee0642_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-xvector-0.46.0-r44h15a9599_2.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-zlibbioc-1.52.0-r44h3df3fcb_2.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/biopython-1.86-py311h49ec1c0_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/blast-2.17.0-h66d330f_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/boltons-25.0.0-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/boost-cpp-1.85.0-h3c6214e_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bowtie2-2.5.4-he96a11b_6.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/brotli-1.2.0-h41a2e66_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/brotli-bin-1.2.0-hf2c8021_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/brotli-python-1.2.0-py311h7c6b74e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/busco-6.0.0-pyhdfd78af_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/bwidget-1.10.1-ha770c72_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/bzip2-1.0.8-hda65f42_8.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/c-ares-1.34.5-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/ca-certificates-2025.11.12-hbd8a1cb_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/cairo-1.18.4-h3394656_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/cd-hit-4.8.1-h5ca1c30_13.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/cdbtools-0.99-h077b44d_12.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/certifi-2025.11.12-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/cffi-2.0.0-py311h03d9500_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/charset-normalizer-3.4.4-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/clustalw-2.1-h9948957_12.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/colorama-0.4.6-pyhd8ed1ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/conda-25.9.1-py311h38be061_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/conda-libmamba-solver-25.4.0-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/conda-package-handling-2.4.0-pyh7900ff3_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/conda-package-streaming-0.12.0-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/contourpy-1.3.3-py311hdf67eae_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/coreutils-9.5-hd590300_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/cpp-expected-1.3.1-h171cf75_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/curl-8.17.0-h4e3cde8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/cycler-0.12.1-pyhd8ed1ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/cyrus-sasl-2.1.28-hd9c7081_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/dbus-1.16.2-h3c4dab8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/dendropy-5.0.8-pyhdfd78af_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/diamond-2.1.16-h13889ed_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/distro-1.9.0-pyhd8ed1ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/dnspython-2.8.0-pyhcf101f3_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/double-conversion-3.3.1-h5888daf_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/entrez-direct-24.0-he881be0_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/ete3-3.1.3-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/famsa-2.4.1-h9ee0642_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/fastme-2.1.6.3-h7b50bb2_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/fastp-1.0.1-heae3180_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/fasttree-2.2.0-h7b50bb2_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/fmt-11.2.0-h07f6e7f_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/font-ttf-dejavu-sans-mono-2.37-hab24e00_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/font-ttf-inconsolata-3.000-h77eed37_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/font-ttf-source-code-pro-2.038-h77eed37_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/font-ttf-ubuntu-0.83-h77eed37_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/fontconfig-2.15.0-h7e30c49_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/fonts-conda-ecosystem-1-0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/fonts-conda-forge-1-hc364b38_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/fonttools-4.60.1-py311h3778330_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/freetype-2.14.1-ha770c72_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/fribidi-1.0.16-hb03c661_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/frozendict-2.4.7-py311h49ec1c0_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gawk-5.3.1-hcd3d067_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gcc_impl_linux-64-15.2.0-hcacfade_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gettext-0.25.1-h3f43e3d_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gettext-tools-0.25.1-h3f43e3d_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gfortran_impl_linux-64-15.2.0-h1b0a18f_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/giflib-5.2.2-hd590300_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/git-2.51.2-pl5321h28be001_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/glib-2.86.1-hbcf1ec1_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/glib-tools-2.86.1-hf516916_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/glpk-5.0-h445213a_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gmp-6.3.0-hac33072_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/graphite2-1.3.14-hecca717_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gsl-2.7-he838d99_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gst-plugins-base-1.24.11-h651a532_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gstreamer-1.24.11-hc37bda9_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gxx_impl_linux-64-15.2.0-h54ccb8d_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/h2-4.3.0-pyhcf101f3_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/harfbuzz-12.2.0-h15599e2_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/hdf5-1.14.3-nompi_h2d575fe_109.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/hisat2-2.2.1-h503566f_8.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/hmmer-3.4-hb6cb901_4.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/hpack-4.1.0-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/htslib-1.22.1-h566b1c6_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/hyperframe-6.1.0-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/icu-75.1-he02047a_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/idna-3.11-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/iqtree-3.0.1-h503566f_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/isa-l-2.31.1-hb9d3cd8_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/joblib-1.5.2-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/jq-1.8.1-h73b1eb8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/jsoncpp-1.9.6-hf42df4d_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/jsonpatch-1.33-pyhd8ed1ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/jsonpointer-3.0.0-py311h38be061_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/julia-1.12.1-h212faf0_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/kallisto-0.51.1-h2b92561_2.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/kernel-headers_linux-64-4.18.0-he073ed8_8.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/keyutils-1.6.3-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/kiwisolver-1.4.9-py311h724c32c_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/kmer-jellyfish-2.3.1-py311pl5321he264feb_6.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/krb5-1.21.3-h659f571_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lame-3.100-h166bdaf_1003.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lcms2-2.17-h717163a_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/ld_impl_linux-64-2.45-h1aa0949_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lerc-4.0.0-h0aef613_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libaec-1.1.4-h3f801dc_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libamd-3.3.3-h456b2da_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libarchive-3.8.1-gpl_h98cc613_100.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libasprintf-0.25.1-h3f43e3d_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libasprintf-devel-0.25.1-h3f43e3d_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libblas-3.9.0-38_h4a7cf45_openblas.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libboost-1.85.0-h0ccab89_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libboost-devel-1.85.0-h00ab1b0_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libboost-headers-1.85.0-ha770c72_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libbrotlicommon-1.2.0-h09219d5_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libbrotlidec-1.2.0-hd53d788_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libbrotlienc-1.2.0-h02bd7ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libbtf-2.3.2-hf02c80a_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcamd-3.3.3-hf02c80a_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcap-2.77-h3ff7636_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcblas-3.9.0-38_h0358290_openblas.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libccolamd-3.3.4-hf02c80a_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcholmod-5.3.1-h9cf07ce_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libclang-cpp20.1-20.1.8-default_h99862b1_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libclang-cpp21.1-21.1.0-default_h99862b1_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libclang13-21.1.0-default_h746c552_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcolamd-3.3.4-hf02c80a_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcups-2.3.3-hb8b1518_5.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcurl-8.17.0-h4e3cde8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcxsparse-4.4.1-hf02c80a_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libdb-6.2.32-h9c3ff4c_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libdeflate-1.25-h17f619e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libdrm-2.4.125-hb03c661_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libedit-3.1.20250104-pl5321h7949ede_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libegl-1.7.0-ha4b6fd6_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libev-4.33-hd590300_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libevent-2.1.12-hf998b51_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libexpat-2.7.1-hecca717_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libffi-3.5.2-h9ec8514_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libflac-1.4.3-h59595ed_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libfreetype-2.14.1-ha770c72_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libfreetype6-2.14.1-h73754d4_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgcc-15.2.0-h767d61c_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/libgcc-devel_linux-64-15.2.0-h73f6952_107.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgcc-ng-15.2.0-h69a702a_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgettextpo-0.25.1-h3f43e3d_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgettextpo-devel-0.25.1-h3f43e3d_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgfortran-15.2.0-h69a702a_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgfortran-ng-15.2.0-h69a702a_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgfortran5-15.2.0-hcd61629_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgit2-1.9.1-h20a291d_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgl-1.7.0-ha4b6fd6_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libglib-2.86.1-h32235b2_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libglvnd-1.7.0-ha4b6fd6_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libglx-1.7.0-ha4b6fd6_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgomp-15.2.0-h767d61c_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libhwloc-2.12.1-default_h3d81e11_1000.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libiconv-1.18-h3b78370_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libidn2-2.3.8-hfac485b_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libjemalloc-5.3.0-h5888daf_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libjpeg-turbo-3.1.2-hb03c661_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libklu-2.3.5-h95ff59c_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/liblapack-3.9.0-38_h47877c9_openblas.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libldl-3.3.2-hf02c80a_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libllvm20-20.1.8-hecd9e04_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libllvm21-21.1.0-hecd9e04_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/liblzma-5.8.1-hb9d3cd8_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/liblzma-devel-5.8.1-hb9d3cd8_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libmamba-2.3.2-hae34dd5_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libmambapy-2.3.2-py311h52fc1f4_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libnghttp2-1.67.0-had1ee68_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libnsl-2.0.1-hb9d3cd8_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libntlm-1.8-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libogg-1.3.5-hd0c01bc_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libopenblas-0.3.30-pthreads_h94d23a6_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libopenblas-ilp64-0.3.30-pthreads_h3e26593_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libopengl-1.7.0-ha4b6fd6_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libopenlibm4-0.8.1-hd590300_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libopenssl-static-3.6.0-hb03c661_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libopus-1.5.2-hd0c01bc_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libparu-1.0.0-hc6afc67_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libpciaccess-0.18-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libpng-1.6.50-h421ea60_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libpq-17.6-h3675c94_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/librbio-4.3.4-hf02c80a_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libsanitizer-15.2.0-hb13aed2_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libsndfile-1.2.2-hc60ed4a_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libsolv-0.7.35-h9463b59_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libspex-3.2.3-h9226d62_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libspqr-4.3.4-h23b7119_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libsqlite-3.51.0-hee844dc_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libssh2-1.11.1-hcf80075_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libstdcxx-15.2.0-h8f9b012_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/libstdcxx-devel_linux-64-15.2.0-h73f6952_107.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libstdcxx-ng-15.2.0-h4852527_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libsuitesparseconfig-7.10.1-h901830b_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libsystemd0-257.10-hd0affe5_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libtiff-4.7.1-h9d88235_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libumfpack-6.3.5-h873dde6_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libunistring-0.9.10-h7f98852_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libunwind-1.6.2-h9c3ff4c_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libutf8proc-2.11.0-hb04c3b8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libuuid-2.41.2-he9a06e4_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libvorbis-1.3.7-h54a6638_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libwebp-base-1.6.0-hd42ef1d_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libxcb-1.17.0-h8a09558_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libxcrypt-4.4.36-hd590300_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libxkbcommon-1.11.0-he8b52b9_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libxml2-2.13.9-h04c0eec_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libxslt-1.1.43-h7a3aeb2_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libzlib-1.3.1-hb9d3cd8_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lp_solve-5.5.2.11-hd590300_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lxml-6.0.2-py311hc53b721_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lz4-c-1.10.0-h5888daf_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lzo-2.10-h280c20c_1002.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/macse-2.07-hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/mafft-7.526-h4bc722e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/make-4.4.1-hb9d3cd8_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/markdown-it-py-4.0.0-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/matplotlib-3.10.8-py311h38be061_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/matplotlib-base-3.10.8-py311h0f3be63_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/maven-3.9.11-ha770c72_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/mcl-22.282-pl5321h7b50bb2_4.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/mdurl-0.1.2-pyhd8ed1ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/menuinst-2.4.1-py311h38be061_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/metaeuk-7.bba0d80-pl5321hd6d6fdc_2.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/metis-5.1.0-hd0bcaf9_1007.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/miniprot-0.18-h577a1d6_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/mmseqs2-18.8cc5c-hd6d6fdc_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/modeltest-ng-0.1.7-hf316886_3.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/mpfr-4.2.1-h90cbb55_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/mpg123-1.32.9-hc50e24c_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/mpi-1.0-openmpi.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/munkres-1.1.4-pyhd8ed1ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/muscle-3.8.1551-h9948957_9.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/mysql-connector-c-6.1.11-h659d440_1008.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/ncbi-vdb-3.2.1-h9948957_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/ncurses-6.5-h2d0b736_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/nlohmann_json-abi-3.12.0-h0f90c79_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/nspr-4.38-h29cc59b_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/nss-3.117-h445c969_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/numpy-2.3.4-py311h2e04523_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/oniguruma-6.9.10-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openblas-ilp64-0.3.30-pthreads_h3d04fff_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openjdk-25.0.1-h5755bd7_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openjpeg-2.5.4-h55fea9a_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openldap-2.6.10-he970967_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openlibm-0.8.1-hd590300_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openmpi-4.1.6-hc5af2df_101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openssl-3.6.0-h26f9b46_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/orthofinder-3.1.0-hdfd78af_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/ossuuid-1.6.2-h5888daf_1001.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/p7zip-16.02-h9c3ff4c_1001.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/packaging-25.0-pyh29332c3_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/pal2nal-14.1-pl5321hdfd78af_3.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pandas-2.3.3-py311hed34c8f_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pandoc-3.8.2.1-ha770c72_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pango-1.56.4-hadf4263_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/pasta-1.9.3-py311hefa8cab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pcre-8.45-h9c3ff4c_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pcre2-10.46-h1321c63_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-5.32.1-7_hd590300_perl5.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/perl-alien-build-2.84-pl5321h7b50bb2_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/perl-alien-libxml2-0.17-pl5321h577a1d6_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-app-cpanminus-1.7048-pl5321hd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-archive-tar-3.04-pl5321hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-business-isbn-3.007-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-business-isbn-data-20210112.006-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-capture-tiny-0.48-pl5321ha770c72_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-carp-1.50-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-class-method-modifiers-2.13-pl5321ha770c72_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-common-sense-3.75-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-compress-raw-bzip2-2.214-pl5321hda65f42_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-compress-raw-zlib-2.214-pl5321h4dac143_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-constant-1.33-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-db_file-1.858-pl5321hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-dbi-1.647-pl5321hb03c661_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-encode-3.21-pl5321hb9d3cd8_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-exporter-5.74-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-exporter-tiny-1.002002-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-extutils-makemaker-7.70-pl5321hd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-ffi-checklib-0.28-pl5321hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-file-chdir-0.1011-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-file-path-2.18-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-file-temp-0.2304-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-file-which-1.24-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-getopt-long-2.58-pl5321hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-importer-0.026-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-inc-latest-0.500-pl5321ha770c72_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/perl-io-compress-2.213-pl5321h503566f_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-io-zlib-1.15-pl5321hdfd78af_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-json-4.10-pl5321hdfd78af_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/perl-json-xs-4.04-pl5321h9948957_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-list-moreutils-0.430-pl5321hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/perl-list-moreutils-xs-0.430-pl5321h7b50bb2_5.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-module-build-0.4234-pl5321ha770c72_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-moo-2.005004-pl5321ha770c72_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-parallel-forkmanager-2.04-pl5321hdfd78af_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-parent-0.243-pl5321hd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-path-tiny-0.124-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-pathtools-3.75-pl5321hb9d3cd8_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-role-tiny-2.002004-pl5321ha770c72_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-scalar-list-utils-1.70-pl5321hb03c661_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-scope-guard-0.21-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-storable-3.15-pl5321hb9d3cd8_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-sub-info-0.002-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-sub-quote-2.006006-pl5321ha770c72_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-term-table-0.025-pl5321hdfd78af_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-test-fatal-0.016-pl5321ha770c72_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-test-warnings-0.031-pl5321ha770c72_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-test2-suite-0.000163-pl5321hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-try-tiny-0.31-pl5321ha770c72_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-types-serialiser-1.01-pl5321hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-uri-5.34-pl5321ha770c72_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/perl-xml-libxml-2.0210-pl5321hf886d80_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-xml-namespacesupport-1.12-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-xml-sax-1.02-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-xml-sax-base-1.09-pl5321hd8ed1ab_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-yaml-1.30-pl5321hdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pillow-12.0.0-py311h07c5bb8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pip-25.3-pyh8b19718_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pixman-0.46.4-h54a6638_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/platformdirs-4.5.0-pyhcf101f3_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pluggy-1.6.0-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/ply-3.11-pyhd8ed1ab_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/pplacer-1.1.alpha19-h9ee0642_2.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/prank-170427-h9948957_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/prodigal-2.6.3-h577a1d6_11.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pthread-stubs-0.4-hb9d3cd8_1002.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pulseaudio-client-17.0-h9a8bead_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pybind11-abi-4-hd8ed1ab_3.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pycosat-0.6.6-py311h49ec1c0_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pycparser-2.22-pyh29332c3_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pygments-2.19.2-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pygtrie-2.5.0-pyhd8ed1ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pymongo-4.15.4-py311h1ddb823_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pyparsing-3.2.5-pyhcf101f3_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pyqt-5.15.11-py311h0580839_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pyqt5-sip-12.17.0-py311h1ddb823_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pyside6-6.9.2-py311h72d58bf_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pysocks-1.7.1-pyha55dd90_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/python-3.11.14-hd63d673_2_cpython.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/python-tzdata-2025.2-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/python_abi-3.11-8_cp311.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pytz-2025.2-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pyyaml-6.0.3-py311h3778330_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/qhull-2020.2-h434a139_5.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/qt-main-5.15.15-h3a7ef08_5.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/qt6-main-6.9.2-h5bd77bc_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-abind-1.4_8-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-amap-0.8_20-r44ha36cffa_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-ape-5.8_1-r44h3704496_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-argparse-2.3.1-r44hc72bb7e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-askpass-1.2.1-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-assertthat-0.2.1-r44hc72bb7e_6.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-backports-1.5.0-r44h54b55ab_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-base-4.4.3-hc038350_5.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-base64enc-0.1_3-r44h54b55ab_1008.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-bh-1.87.0_1-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-biasedurn-2.0.12-r44h3697838_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-bit-4.6.0-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-bit64-4.6.0_1-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-bitops-1.0_9-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-blob-1.2.4-r44hc72bb7e_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-bms-0.3.5-r44hc72bb7e_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-brew-1.0_10-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-broom-1.0.10-r44hc72bb7e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-bslib-0.9.0-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-cachem-1.1.0-r44h54b55ab_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-callr-3.7.6-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-catools-1.18.3-r44h3697838_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-cellranger-1.1.0-r44hc72bb7e_1008.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-checkmate-2.3.3-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-cli-3.6.5-r44h3697838_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-clipr-0.8.0-r44hc72bb7e_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-cluster-2.1.8.1-r44heaba542_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-codetools-0.2_20-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-collections-0.3.9-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-colorspace-2.1_2-r44h54b55ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-commonmark-2.0.0-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-conflicted-1.2.0-r44h785f33e_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-cpp11-0.5.2-r44h785f33e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-crayon-1.5.3-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-curl-7.0.0-r44h10955f1_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-cyclocomp-1.1.1-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-data.table-1.17.8-r44h1c8cec4_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-dbi-1.2.3-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-dbplyr-2.5.1-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-desc-1.4.3-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-digest-0.6.38-r44h3697838_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-distributional-0.5.0-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-dplyr-1.1.4-r44h3697838_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-dtplyr-1.3.2-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-ellipsis-0.3.2-r44h54b55ab_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-evaluate-1.0.5-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-fansi-1.0.6-r44h54b55ab_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-farver-2.1.2-r44h3697838_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-fastcluster-1.3.0-r44h3697838_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-fastmap-1.2.0-r44h3697838_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-fastmatch-1.1_6-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-filelock-1.0.3-r44h54b55ab_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-findpython-1.0.9-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-fontawesome-0.5.3-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-forcats-1.0.1-r44hc72bb7e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-formatr-1.14-r44hc72bb7e_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-fs-1.6.6-r44h3697838_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-futile.logger-1.4.3-r44hc72bb7e_1007.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-futile.options-1.0.1-r44hc72bb7e_1006.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-gargle-1.6.0-r44h785f33e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-generics-0.1.4-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-ggplot2-4.0.1-r44h785f33e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-glue-1.8.0-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-googledrive-2.1.2-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-googlesheets4-1.1.2-r44h785f33e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-gplots-3.2.0-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-gridextra-2.3-r44hc72bb7e_1007.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-gtable-0.3.6-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-gtools-3.9.5-r44h54b55ab_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-haven-2.5.5-r44h6d565e7_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-highr-0.11-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-hms-1.1.4-r44hc72bb7e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-htmltools-0.5.8.1-r44h3697838_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-httr-1.4.7-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-httr2-1.2.1-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-hwriter-1.3.2.1-r44hc72bb7e_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-ids-1.0.1-r44hc72bb7e_5.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-igraph-2.1.4-r44hadbbdbc_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-inline-0.3.21-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-isoband-0.2.7-r44h3697838_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-jquerylib-0.1.4-r44hc72bb7e_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-jsonlite-2.0.0-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-kernsmooth-2.23_26-r44ha0a88a1_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-knitr-1.50-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-labeling-0.4.3-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-lambda.r-1.2.4-r44hc72bb7e_5.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-languageserver-0.3.16-r44h54b55ab_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-lattice-0.22_7-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-lazyeval-0.2.2-r44h54b55ab_6.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-lifecycle-1.0.4-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-lintr-3.2.0-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-locfit-1.5_9.12-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-loo-2.8.0-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-lubridate-1.9.4-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-magrittr-2.0.4-r44h54b55ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-matrix-1.7_4-r44h0e4624f_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-matrixstats-1.5.0-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-memoise-2.0.1-r44hc72bb7e_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-mgcv-1.9_4-r44h0e4624f_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-mime-0.13-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-modelr-0.1.11-r44hc72bb7e_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-munsell-0.5.1-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-nlme-3.1_168-r44heaba542_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-numderiv-2016.8_1.1-r44hc72bb7e_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-openssl-2.3.4-r44h50f7d53_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-phangorn-2.12.1-r44hf1899b2_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-pillar-1.11.1-r44hc72bb7e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-pkgbuild-1.4.8-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-pkgconfig-2.0.3-r44hc72bb7e_5.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-pkgload-1.4.1-r44hc72bb7e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-plogr-0.2.0-r44hc72bb7e_1007.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-plyr-1.8.9-r44h3697838_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-png-0.1_8-r44h6b2d295_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-posterior-1.6.1-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-prettyunits-1.2.0-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-processx-3.8.6-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-progress-1.2.3-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-ps-1.9.1-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-purrr-1.2.0-r44h54b55ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-quadprog-1.5_8-r44ha0a88a1_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-quickjsr-1.8.1-r44h3697838_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-r.cache-0.17.0-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-r.methodss3-1.8.2-r44hc72bb7e_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-r.oo-1.27.1-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-r.utils-2.13.0-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-r6-2.6.1-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-ragg-1.5.0-r44h9f1dc4d_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rappdirs-0.3.3-r44h54b55ab_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rcolorbrewer-1.1_3-r44h785f33e_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rcpp-1.1.0-r44h3697838_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rcpparmadillo-15.0.2_2-r44h3704496_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rcppeigen-0.3.4.0.2-r44h3704496_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rcppparallel-5.1.11_1-r44hbd9b9cf_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rcurl-1.98_1.17-r44hb79926e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-readr-2.1.5-r44h3697838_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-readxl-1.4.5-r44h10e25cc_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rematch-2.0.0-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rematch2-2.1.2-r44hc72bb7e_5.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-remotes-2.5.0-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-repr-1.1.7-r44h785f33e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-reprex-2.1.1-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-reshape2-1.4.5-r44h3697838_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/r-restfulr-0.0.16-r44h5ef9028_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rex-1.2.1-r44hc72bb7e_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rjson-0.2.23-r44h3697838_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rlang-1.1.6-r44h3697838_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rmarkdown-2.30-r44hc72bb7e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-roxygen2-7.3.3-r44h3697838_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rprojroot-2.1.1-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rsqlite-2.4.4-r44h3697838_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rstan-2.32.7-r44h3697838_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rstudioapi-0.17.1-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rvest-1.0.5-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-s7-0.2.0-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-sass-0.4.10-r44h3697838_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-scales-1.4.0-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-selectr-0.4_2-r44hc72bb7e_5.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-sm-2.2_6.0-r44heaba542_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-snow-0.4_4-r44hc72bb7e_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-stanheaders-2.32.10-r44ha36cffa_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-statmod-1.5.1-r44hb1d0f04_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-stringi-1.8.7-r44h2dae267_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-stringr-1.6.0-r44h785f33e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-styler-1.11.0-r44hc72bb7e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-survival-3.8_3-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-sys-3.4.3-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-systemfonts-1.3.1-r44h74f4acd_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-tensora-0.36.2.1-r44h54b55ab_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-textshaping-1.0.4-r44h74f4acd_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-tibble-3.3.0-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-tidyr-1.3.1-r44h3697838_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-tidyselect-1.2.1-r44hc72bb7e_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-tidyverse-2.0.0-r44h785f33e_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-timechange-0.3.0-r44h3697838_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-tinytex-0.57-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-tzdb-0.5.0-r44h3697838_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-utf8-1.2.6-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-uuid-1.2_1-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-vctrs-0.6.5-r44h3697838_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-vioplot-0.5.1-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-viridislite-0.4.2-r44hc72bb7e_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-vroom-1.6.6-r44h3697838_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-withr-3.0.2-r44hc72bb7e_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-xfun-0.54-r44h3697838_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-xml-3.99_0.17-r44h7c9d5c0_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-xml2-1.4.0-r44hc6fd541_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-xmlparsedata-1.0.5-r44hc72bb7e_4.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-xtable-1.8_4-r44hc72bb7e_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-yaml-2.3.10-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-zoo-1.8_14-r44h54b55ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/raxml-8.2.13-h7b50bb2_3.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/raxml-ng-1.2.2-h6747034_2.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/readline-8.2-h8c095d6_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/reproc-14.2.5.post0-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/reproc-cpp-14.2.5.post0-h5888daf_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/requests-2.32.5-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/rich-14.2.0-pyhcf101f3_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/ruamel.yaml-0.18.16-py311h49ec1c0_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/ruamel.yaml.clib-0.2.14-py311h49ec1c0_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/salmon-1.10.3-haf24da9_3.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/samtools-1.22.1-h96c455f_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/scikit-learn-1.7.2-py311hc3e1efb_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/scipy-1.16.3-py311h1e13796_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/sed-4.9-h6688a6e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/sepp-4.5.6-py311haab0aaa_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/seqkit-2.10.1-he881be0_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/seqmagick-0.8.6-pyhdfd78af_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/setuptools-80.9.0-pyhff2d567_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/simdjson-4.0.7-hb700be7_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/sip-6.10.0-py311h1ddb823_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/six-1.17.0-pyhe01879c_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/sniffio-1.3.1-pyhd8ed1ab_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/sqlite-3.51.0-heff268d_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/sra-tools-3.2.1-h4304569_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/suitesparse-7.10.1-h5b2951e_7100101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/sysroot_linux-64-2.28-h4ee821c_8.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/tar-1.35-h3b78370_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/tbb-2022.3.0-h8d10470_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/tbb-devel-2022.3.0-h74b38a2_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/threadpoolctl-3.6.0-pyhecae5ae_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/tk-8.6.13-noxft_ha0e22de_103.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/tktable-2.10-h8d826fa_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/toml-0.10.2-pyhd8ed1ab_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/tomli-2.3.0-pyhcf101f3_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/tomlkit-0.13.3-pyha770c72_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/tornado-6.5.2-py311h49ec1c0_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/tqdm-4.67.1-pyhd8ed1ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/transdecoder-5.7.1-pl5321hdfd78af_2.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/treeshrink-1.3.9-pyhdfd78af_1.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/trimal-1.5.0-h9948957_2.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/trimmomatic-0.40-hdfd78af_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/trinity-2.15.2-pl5321h077b44d_6.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/truststore-0.10.3-pyhe01879c_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/typing_extensions-4.15.0-pyhcf101f3_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/tzdata-2025b-h78e105d_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/ucsc-fatotwobit-482-hdc0a859_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/ucsc-twobitinfo-482-hdc0a859_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/unicodedata2-17.0.0-py311h49ec1c0_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/urllib3-2.5.0-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/wayland-1.24.0-hd6090a7_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/wget-1.21.4-hda4d442_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/wheel-0.45.1-pyhd8ed1ab_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xcb-util-0.4.1-h4f16b4b_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xcb-util-cursor-0.1.5-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xcb-util-image-0.4.0-hb711507_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xcb-util-keysyms-0.4.1-hb711507_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xcb-util-renderutil-0.3.10-hb711507_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xcb-util-wm-0.4.2-hb711507_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xkeyboard-config-2.46-hb03c661_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/xmltodict-1.0.2-pyhcf101f3_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libice-1.1.2-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libsm-1.2.6-he73a12e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libx11-1.8.12-h4f16b4b_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxau-1.0.12-hb03c661_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxcomposite-0.4.6-hb9d3cd8_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxcursor-1.2.3-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxdamage-1.1.6-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxdmcp-1.1.5-hb03c661_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxext-1.3.6-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxfixes-6.0.2-hb03c661_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxi-1.8.2-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxrandr-1.5.4-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxrender-0.9.12-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxshmfence-1.3.3-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxt-1.3.1-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxtst-1.2.5-hb9d3cd8_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxxf86vm-1.1.6-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-xextproto-7.3.0-hb9d3cd8_1004.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/xvfbwrapper-0.2.15-pyhd8ed1ab_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xz-5.8.1-hbcc6ac9_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xz-gpl-tools-5.8.1-hbcc6ac9_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xz-tools-5.8.1-hb9d3cd8_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/yaml-0.2.5-h280c20c_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/yaml-cpp-0.8.0-h3f2d84a_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/yq-3.4.3-pyhe01879c_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/zlib-1.3.1-hb9d3cd8_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/zlib-ng-2.2.5-hde8ca8f_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/zstandard-0.25.0-py311haee01d2_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/zstd-1.5.7-hb8e6e7a_2.conda + mrbayes: + channels: + - url: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/ + - url: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/ + - url: https://conda.anaconda.org/conda-forge/ + - url: https://conda.anaconda.org/bioconda/ + packages: + linux-64: + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/_libgcc_mutex-0.1-conda_forge.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/_openmp_mutex-4.5-2_gnu.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/alsa-lib-1.2.14-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/beagle-lib-3.1.2-h503566f_5.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/bzip2-1.0.8-hda65f42_8.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/ca-certificates-2025.11.12-hbd8a1cb_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/cairo-1.18.4-h3394656_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/font-ttf-dejavu-sans-mono-2.37-hab24e00_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/font-ttf-inconsolata-3.000-h77eed37_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/font-ttf-source-code-pro-2.038-h77eed37_0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/font-ttf-ubuntu-0.83-h77eed37_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/fontconfig-2.15.0-h7e30c49_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/fonts-conda-ecosystem-1-0.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/fonts-conda-forge-1-hc364b38_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/freetype-2.14.1-ha770c72_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/giflib-5.2.2-hd590300_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/graphite2-1.3.14-hecca717_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/harfbuzz-12.2.0-h15599e2_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/icu-75.1-he02047a_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/keyutils-1.6.3-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/krb5-1.21.3-h659f571_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lcms2-2.17-h717163a_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lerc-4.0.0-h0aef613_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcups-2.3.3-hb8b1518_5.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libdeflate-1.25-h17f619e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libedit-3.1.20250104-pl5321h7949ede_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libexpat-2.7.1-hecca717_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libffi-3.5.2-h9ec8514_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libfreetype-2.14.1-ha770c72_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libfreetype6-2.14.1-h73754d4_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgcc-15.2.0-h767d61c_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgcc-ng-15.2.0-h69a702a_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgfortran-15.2.0-h69a702a_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgfortran-ng-15.2.0-h69a702a_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgfortran5-15.2.0-hcd61629_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libglib-2.86.1-h32235b2_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgomp-15.2.0-h767d61c_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libiconv-1.18-h3b78370_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libjpeg-turbo-3.1.2-hb03c661_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libltdl-2.4.3a-h5888daf_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/liblzma-5.8.1-hb9d3cd8_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libpng-1.6.50-h421ea60_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libstdcxx-15.2.0-h8f9b012_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libstdcxx-ng-15.2.0-h4852527_7.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libtiff-4.7.1-h9d88235_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libtool-2.5.4-h5888daf_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libuuid-2.41.2-he9a06e4_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libwebp-base-1.6.0-hd42ef1d_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libxcb-1.17.0-h8a09558_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libzlib-1.3.1-hb9d3cd8_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/mpi-1.0-openmpi.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/mrbayes-3.2.7-hd0d793b_7.tar.bz2 + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/ncurses-6.5-h2d0b736_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openjdk-25.0.1-h5755bd7_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openmpi-4.1.6-hc5af2df_101.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openssl-3.6.0-h26f9b46_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pcre2-10.46-h1321c63_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pixman-0.46.4-h54a6638_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pthread-stubs-0.4-hb9d3cd8_1002.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/readline-8.2-h8c095d6_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libice-1.1.2-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libsm-1.2.6-he73a12e_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libx11-1.8.12-h4f16b4b_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxau-1.0.12-hb03c661_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxdmcp-1.1.5-hb03c661_1.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxext-1.3.6-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxfixes-6.0.2-hb03c661_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxi-1.8.2-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxrandr-1.5.4-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxrender-0.9.12-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxt-1.3.1-hb9d3cd8_0.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxtst-1.2.5-hb9d3cd8_3.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/zlib-1.3.1-hb9d3cd8_2.conda + - conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/zstd-1.5.7-hb8e6e7a_2.conda +packages: +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/_libgcc_mutex-0.1-conda_forge.tar.bz2 + sha256: fe51de6107f9edc7aa4f786a70f4a883943bc9d39b3bb7307c04c41410990726 + md5: d7c89558ba9fa0495403155b64376d81 + license: None + size: 2562 + timestamp: 1578324546067 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/_openmp_mutex-4.5-2_gnu.tar.bz2 + build_number: 16 + sha256: fbe2c5e56a653bebb982eda4876a9178aedfc2b545f25d0ce9c4c0b508253d22 + md5: 73aaf86a425cc6e73fcf236a5a46396d + depends: + - _libgcc_mutex 0.1 conda_forge + - libgomp >=7.5.0 + constrains: + - openmp_impl 9999 + license: BSD-3-Clause + license_family: BSD + size: 23621 + timestamp: 1650670423406 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/_r-mutex-1.0.1-anacondar_1.tar.bz2 + sha256: e58f9eeb416b92b550e824bcb1b9fb1958dee69abfe3089dfd1a9173e3a0528a + md5: 19f9db5f4f1b7f5ef5f6d67207f25f38 + license: BSD + size: 3566 + timestamp: 1562343890778 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/alsa-lib-1.2.14-hb9d3cd8_0.conda + sha256: b9214bc17e89bf2b691fad50d952b7f029f6148f4ac4fe7c60c08f093efdf745 + md5: 76df83c2a9035c54df5d04ff81bcc02d + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: LGPL-2.1-or-later + license_family: GPL + size: 566531 + timestamp: 1744668655747 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/archspec-0.2.5-pyhd8ed1ab_0.conda + sha256: eb68e1ce9e9a148168a4b1e257a8feebffdb0664b557bb526a1e4853f2d2fc00 + md5: 845b38297fca2f2d18a29748e2ece7fa + depends: + - python >=3.9 + license: MIT OR Apache-2.0 + size: 50894 + timestamp: 1737352715041 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/argcomplete-3.6.3-pyhd8ed1ab_0.conda + sha256: a2a1879c53b7a8438c898d20fa5f6274e4b1c30161f93b7818236e9df6adffde + md5: 8f37c8fb7116a18da04e52fa9e2c8df9 + depends: + - python >=3.10 + license: Apache-2.0 + license_family: Apache + size: 42386 + timestamp: 1760975036972 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/aria2-1.37.0-hbc8128a_2.conda + sha256: 06ac389ee45049af40aeb9940eacef92f04d6b5741fc1154be282f420479a49f + md5: 03b8874fa70df577f3eee53085d025cf + depends: + - c-ares >=1.28.1,<2.0a0 + - libgcc-ng >=12 + - libsqlite >=3.46.0,<4.0a0 + - libssh2 >=1.11.0,<2.0a0 + - libstdcxx-ng >=12 + - libxml2 >=2.12.7,<2.14.0a0 + - libzlib >=1.3.1,<2.0a0 + - openssl >=3.3.1,<4.0a0 + license: GPL-2.0-only + license_family: GPL + size: 1638055 + timestamp: 1718840932941 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/arpack-3.9.1-nompi_hf03ea27_102.conda + sha256: 6d71343420292132be0192ddd962b308f7b8a0a0630d1db83fb9d65e8167c6ce + md5: e09af397232ef1070e0b6cbf4c64aacb + depends: + - __glibc >=2.17,<3.0.a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - libgfortran + - libgfortran5 >=13.3.0 + - liblapack >=3.9.0,<4.0a0 + - libstdcxx >=13 + license: BSD-3-Clause + license_family: BSD + size: 130412 + timestamp: 1736083992796 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/aster-1.23-h9948957_0.tar.bz2 + sha256: 88edc4b9e475a55befe44c144982535ca59b484d127b1ebefdccf3a6443a8d62 + md5: a670e064ff18d1b44deac82e3a00ae8b + depends: + - libgcc >=13 + - libstdcxx >=13 + license: AGPL-3.0-or-later + license_family: AGPL + size: 1585973 + timestamp: 1753126202062 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/attr-2.5.2-h39aace5_0.conda + sha256: a9c114cbfeda42a226e2db1809a538929d2f118ef855372293bd188f71711c48 + md5: 791365c5f65975051e4e017b5da3abf5 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: GPL-2.0-or-later + license_family: GPL + size: 68072 + timestamp: 1756738968573 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/augustus-3.5.0-pl5321h57ba348_8.tar.bz2 + sha256: 1ef9e78d3001af0a5389d33b1978d7f46b444b6c87083462b31ed28b5f9d25c3 + md5: c4b20f1ce9c3f1ddaecc5598dbfe1cdf + depends: + - bamtools >=2.5.3,<3.0a0 + - biopython + - boost-cpp + - cdbtools + - diamond + - gsl >=2.7,<2.8.0a0 + - htslib >=1.22,<1.23.0a0 + - libamd >=3.3.3,<4.0a0 + - libblas >=3.9.0,<4.0a0 + - libbtf >=2.3.2,<3.0a0 + - libcamd >=3.3.3,<4.0a0 + - libcblas >=3.9.0,<4.0a0 + - libccolamd >=3.3.4,<4.0a0 + - libcholmod >=5.3.1,<6.0a0 + - libcolamd >=3.3.4,<4.0a0 + - libcxsparse >=4.4.1,<5.0a0 + - libgcc >=13 + - libklu >=2.3.5,<3.0a0 + - libldl >=3.3.2,<4.0a0 + - libparu >=1.0.0,<2.0a0 + - librbio >=4.3.4,<5.0a0 + - libspex >=3.2.3,<4.0a0 + - libspqr >=4.3.4,<5.0a0 + - libsqlite >=3.50.2,<4.0a0 + - libstdcxx >=13 + - libsuitesparseconfig >=7.10.1,<8.0a0 + - libumfpack >=6.3.5,<7.0a0 + - libzlib >=1.3.1,<2.0a0 + - lp_solve + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-app-cpanminus + - perl-dbi + - perl-file-which + - perl-module-build 0.4234.* + - perl-parallel-forkmanager + - perl-scalar-list-utils + - perl-yaml + - samtools >=1.22,<2.0a0 + - sqlite + - suitesparse >=7.10.1,<8.0a0 + - tar + - ucsc-fatotwobit + - ucsc-twobitinfo + license: Artistic License + license_family: Other + size: 33115645 + timestamp: 1751945933327 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bamtools-2.5.3-he132191_0.tar.bz2 + sha256: 18d0c7e69cc6c775184ca7034d8e2f8b6de4a516c3de74792391882ef1bdcd0d + md5: e3302f3cd140ded877f39abe612bed08 + depends: + - jsoncpp >=1.9.6,<1.9.7.0a0 + - libgcc >=13 + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + license: MIT + license_family: MIT + size: 783519 + timestamp: 1747609084710 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bbmap-39.37-he5f24ec_0.conda + sha256: 58c6741ef8a91be4fbfcc56fb731dd63d839fe3e57f0718685a4650df7d03504 + md5: af30349cd8e50dbe5a653153b7acf06a + depends: + - bzip2 >=1.0.8,<2.0a0 + - libgcc >=13 + - openjdk >=11.0.1 + - samtools >=1.22.1,<2.0a0 + license: UC-LBL license (see package) + size: 14175586 + timestamp: 1759800503964 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/beagle-lib-3.1.2-h503566f_5.tar.bz2 + sha256: c989043bf6aada692da9d1bacd961f6bf333ea4328d659262e98be09388add5e + md5: 13cb5591f148d17668f1ea6435f4ab6e + depends: + - libgcc >=13 + - libstdcxx >=13 + - libtool + - openjdk + license: GPL-3.0-or-later + license_family: GPL3 + size: 272751 + timestamp: 1734426483233 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/beagle-lib-4.0.1-h9948957_3.tar.bz2 + sha256: 1858a6b234a92baac967e380daf4eb89ca96e44851dd26f00144e0ef3f56843f + md5: 135e129beaebcf0fba57fd440380c468 + depends: + - libgcc >=13 + - libstdcxx >=13 + license: GPL-3.0-or-later + license_family: GPL3 + size: 1985669 + timestamp: 1748480750171 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/beast-10.5.0-hdfd78af_0.tar.bz2 + sha256: 780da4d9fd2a0507c651d0a1bd6fbd8ac1ca65ca8c69536e636b3fe44fb3d12c + md5: 433688488332192b41de62994f5a45ea + depends: + - beagle-lib + - openjdk + license: LGPL-2.1-or-later + license_family: LGPL + size: 19339740 + timestamp: 1751501790109 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/binutils_impl_linux-64-2.45-h9d8b0ac_0.conda + sha256: 1733bd616f0e7afdc926e4eb80b00483ebdc51bc6aadf7c4b7242ed93044e25b + md5: 0f846eecce9004022f9706252b143b0f + depends: + - ld_impl_linux-64 2.45 h1aa0949_0 + - sysroot_linux-64 + - zstd >=1.5.7,<1.6.0a0 + license: GPL-3.0-only + size: 3781434 + timestamp: 1763060453906 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-annotate-1.84.0-r44hdfd78af_0.tar.bz2 + sha256: b3735b7f02560df4ffb7867c71cea2f2247fcc1020d3c3648bd877acacef99f4 + md5: 507b59c419691df87c71503a47f2c9e0 + depends: + - bioconductor-annotationdbi >=1.68.0,<1.69.0 + - bioconductor-biobase >=2.66.0,<2.67.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - r-base >=4.4,<4.5.0a0 + - r-dbi + - r-httr + - r-xml + - r-xtable + license: Artistic-2.0 + size: 2122564 + timestamp: 1735020419673 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-annotationdbi-1.68.0-r44hdfd78af_0.tar.bz2 + sha256: ab45318a99e061a52db5a06d51e56789651a5ac3bff14813b19404b45976f6ba + md5: cdec31da7826d82f22cdebc22c8755d2 + depends: + - bioconductor-biobase >=2.66.0,<2.67.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-keggrest >=1.46.0,<1.47.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - r-base >=4.4,<4.5.0a0 + - r-dbi + - r-rsqlite + license: Artistic-2.0 + size: 5206121 + timestamp: 1734707761135 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-biobase-2.66.0-r44h3df3fcb_0.tar.bz2 + sha256: 5f482a491f073d6e2441dc192c441f06857eb0039695ef564cea17ffe090d57a + md5: 56f651b4dbe8625ba510c7c3233da8a9 + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + size: 2674289 + timestamp: 1734316615330 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-biocfilecache-2.14.0-r44hdfd78af_0.tar.bz2 + sha256: 53f657d4576994865f7c955e12f57e614101e7c2728bab160c1855d9be353185 + md5: 3f79b6f2157a77fce7eca6cd95082212 + depends: + - r-base >=4.4,<4.5.0a0 + - r-curl + - r-dbi + - r-dbplyr >=1.0.0 + - r-dplyr + - r-filelock + - r-httr + - r-rsqlite + license: Artistic-2.0 + size: 987067 + timestamp: 1734194474836 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-biocgenerics-0.52.0-r44hdfd78af_3.tar.bz2 + sha256: 372f2a4b5f1802dc76347b1e87d8a75fea204304fa5bb3239e54737907761d7d + md5: 8a9defade51c2c2a6b90a4474dcdfdfc + depends: + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + size: 694499 + timestamp: 1738084816476 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-biocio-1.16.0-r44hdfd78af_0.tar.bz2 + sha256: a8358d5d2c008a119ae6a33020f0e97a70097f0b00b52ac273bd4ad3662b9e61 + md5: b243a91f84910e5e92c40ff42161cb0c + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + size: 472561 + timestamp: 1734490335765 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-biocparallel-1.40.0-r44he5774e6_1.tar.bz2 + sha256: b430a140dc4303e07f5ac4ebfeac1f1d3b65822b6d3cedfaa628d331f94403d4 + md5: 09b7ce5c49548c48d102721a0dea6492 + depends: + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - libstdcxx >=13 + - r-base >=4.4,<4.5.0a0 + - r-bh + - r-codetools + - r-cpp11 + - r-futile.logger + - r-snow + license: GPL-2 | GPL-3 + size: 1683869 + timestamp: 1738715593687 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-biomart-2.62.0-r44hdfd78af_0.tar.bz2 + sha256: 876271767753bb85aa02a4f6de7390ec8cafefc478e3b099b6bc4605ef5b831a + md5: 39cfd13060df28086f6c8e4e35ba35d8 + depends: + - bioconductor-annotationdbi >=1.68.0,<1.69.0 + - bioconductor-biocfilecache >=2.14.0,<2.15.0 + - r-base >=4.4,<4.5.0a0 + - r-curl + - r-digest + - r-httr2 + - r-progress + - r-rappdirs + - r-stringr + - r-xml2 + license: Artistic-2.0 + size: 944468 + timestamp: 1734753866060 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-biostrings-2.74.0-r44h3df3fcb_1.tar.bz2 + sha256: 87043f05ff80d5a4556986ee4180fc9ab2c5f5a0be049524177f22865e16f08c + md5: 964a5ef8fe0d0e9a599efab3fa7c3a71 + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0a0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0a0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-iranges >=2.40.0,<2.41.0a0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0a0 + - bioconductor-xvector >=0.46.0,<0.47.0 + - bioconductor-xvector >=0.46.0,<0.47.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - r-base >=4.4,<4.5.0a0 + - r-crayon + license: Artistic-2.0 + size: 14472562 + timestamp: 1737051379092 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-ctc-1.80.0-r44hdfd78af_0.tar.bz2 + sha256: a8a24fefd1e2a6548496d08282fca0326cd2883c0bdde367d03c381052b87a7d + md5: 90305a29953c5dc42d2c2ce1b0a3a6a5 + depends: + - r-amap + - r-base >=4.4,<4.5.0a0 + license: GPL-2 + size: 351401 + timestamp: 1734202056843 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-data-packages-20250625-hdfd78af_0.tar.bz2 + sha256: 1f684de74f51caaab7aeffbc0821d45b5ae86f94f55bcf84ca285f76dd57b310 + md5: 34d7066b99d7e6769305dcebf0a9de87 + depends: + - curl + - r-base + - yq + license: MIT + size: 256159 + timestamp: 1750854850621 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-delayedarray-0.32.0-r44h3df3fcb_1.tar.bz2 + sha256: ba4f80302cdeed5d9158349f5714c71ca26eac36dc9795c2276d47fef27c0195 + md5: 2a69a0cd9896594a301067804e65b8d0 + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0a0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-iranges >=2.40.0,<2.41.0a0 + - bioconductor-matrixgenerics >=1.18.0,<1.19.0 + - bioconductor-matrixgenerics >=1.18.0,<1.19.0a0 + - bioconductor-s4arrays >=1.6.0,<1.7.0 + - bioconductor-s4arrays >=1.6.0,<1.7.0a0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0a0 + - bioconductor-sparsearray >=1.6.0,<1.7.0 + - bioconductor-sparsearray >=1.6.0,<1.7.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - r-base >=4.4,<4.5.0a0 + - r-matrix + license: Artistic-2.0 + size: 2625964 + timestamp: 1738828407347 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-deseq2-1.46.0-r44he5774e6_1.tar.bz2 + sha256: 3246fe045acf88745f70dc0dd34be27fccd4086e6f7528c1e6667838fc5a92c2 + md5: edfea8b9e0555db0870cdcc504b8f032 + depends: + - bioconductor-biobase >=2.66.0,<2.67.0 + - bioconductor-biobase >=2.66.0,<2.67.0a0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0a0 + - bioconductor-biocparallel >=1.40.0,<1.41.0 + - bioconductor-biocparallel >=1.40.0,<1.41.0a0 + - bioconductor-genomicranges >=1.58.0,<1.59.0 + - bioconductor-genomicranges >=1.58.0,<1.59.0a0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-iranges >=2.40.0,<2.41.0a0 + - bioconductor-matrixgenerics >=1.18.0,<1.19.0 + - bioconductor-matrixgenerics >=1.18.0,<1.19.0a0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0a0 + - bioconductor-summarizedexperiment >=1.36.0,<1.37.0 + - bioconductor-summarizedexperiment >=1.36.0,<1.37.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - libstdcxx >=13 + - r-base >=4.4,<4.5.0a0 + - r-ggplot2 >=3.4.0 + - r-locfit + - r-matrixstats + - r-rcpp >=0.11.0 + - r-rcpparmadillo + license: LGPL (>= 3) + size: 3236077 + timestamp: 1744730254165 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-dexseq-1.52.0-r44hdfd78af_0.tar.bz2 + sha256: b0901cfc7c521d5392c32410713510df40600b6d17a29bdfe59a9dd7eea7d582 + md5: b7daa61b3b8db8fee2c58a84c4d97102 + depends: + - bioconductor-annotationdbi >=1.68.0,<1.69.0 + - bioconductor-biobase >=2.66.0,<2.67.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocparallel >=1.40.0,<1.41.0 + - bioconductor-biomart >=2.62.0,<2.63.0 + - bioconductor-deseq2 >=1.46.0,<1.47.0 + - bioconductor-genefilter >=1.88.0,<1.89.0 + - bioconductor-geneplotter >=1.84.0,<1.85.0 + - bioconductor-genomicranges >=1.58.0,<1.59.0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-rsamtools >=2.22.0,<2.23.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-summarizedexperiment >=1.36.0,<1.37.0 + - r-base >=4.4,<4.5.0a0 + - r-hwriter + - r-rcolorbrewer + - r-statmod + - r-stringr + license: GPL (>= 3) + size: 2223736 + timestamp: 1735163849673 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-edger-4.4.0-r44h3df3fcb_0.tar.bz2 + sha256: 2930abe22ca41521a6d09d6b042bd6cf2c99ba55ead766924de37a3c9e60599c + md5: 249af741dc8dac49eb847ac323b1f2ab + depends: + - bioconductor-limma >=3.62.0,<3.63.0 + - bioconductor-limma >=3.62.0,<3.63.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - r-base >=4.4,<4.5.0a0 + - r-locfit + license: GPL (>=2) + size: 2935132 + timestamp: 1734215006388 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-genefilter-1.88.0-r44h81e381d_1.tar.bz2 + sha256: f06d4a75cbec527e19ccd800221dbfc160cfafe9abd9a670245d1cdf03c9fcdf + md5: c28ab340e270bb9804e55bcae072edb7 + depends: + - bioconductor-annotate >=1.84.0,<1.85.0 + - bioconductor-annotate >=1.84.0,<1.85.0a0 + - bioconductor-annotationdbi >=1.68.0,<1.69.0 + - bioconductor-annotationdbi >=1.68.0,<1.69.0a0 + - bioconductor-biobase >=2.66.0,<2.67.0 + - bioconductor-biobase >=2.66.0,<2.67.0a0 + - bioconductor-matrixgenerics >=1.18.0,<1.19.0 + - bioconductor-matrixgenerics >=1.18.0,<1.19.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - libgfortran + - libgfortran5 >=13.3.0 + - liblapack >=3.9.0,<4.0a0 + - libstdcxx >=13 + - r-base >=4.4,<4.5.0a0 + - r-survival + license: Artistic-2.0 + size: 1525961 + timestamp: 1738804069812 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-genelendatabase-1.42.0-r44hdfd78af_0.tar.bz2 + sha256: f8cb2c5fbdc0c492495c907558ab5ddda440a3b937a47da3c8d524c040ba351f + md5: 1618f198361b9e897b11fa092c51af84 + depends: + - bioconductor-data-packages >=20241103 + - bioconductor-genomicfeatures >=1.58.0,<1.59.0 + - bioconductor-rtracklayer >=1.66.0,<1.67.0 + - bioconductor-txdbmaker >=1.2.0,<1.3.0 + - curl + - r-base >=4.4,<4.5.0a0 + license: LGPL (>= 2) + size: 12542 + timestamp: 1735061658097 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-geneplotter-1.84.0-r44hdfd78af_0.tar.bz2 + sha256: 93cc1630f13c4c752e4289a339b6e4695366e071603de4bbac2e6785f056ed13 + md5: 3e0af117beaa027ee105bff96aaaef80 + depends: + - bioconductor-annotate >=1.84.0,<1.85.0 + - bioconductor-annotationdbi >=1.68.0,<1.69.0 + - bioconductor-biobase >=2.66.0,<2.67.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - r-base >=4.4,<4.5.0a0 + - r-lattice + - r-rcolorbrewer + license: Artistic-2.0 + size: 1910336 + timestamp: 1735149345431 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-genomeinfodb-1.42.0-r44hdfd78af_2.tar.bz2 + sha256: 7d7289e06ccb215f3663c96ead00e3a53261fdd2d4ff8b529a78b1a15d6c415a + md5: 52f21aeff5a62bf53a37a4ad4d197e06 + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-genomeinfodbdata >=1.2.0,<1.3.0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-ucsc.utils >=1.2.0,<1.3.0 + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + size: 4413229 + timestamp: 1738086236263 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-genomeinfodbdata-1.2.13-r44hdfd78af_0.tar.bz2 + sha256: ac34a7919b6307dfcdc95c50c3dc5c01838edd43f214329549758fd236775fd9 + md5: 06d453df3bc59956a3ffac7674652f44 + depends: + - bioconductor-data-packages >=20241103 + - curl + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + size: 8473 + timestamp: 1734410193685 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-genomicalignments-1.42.0-r44h3df3fcb_1.tar.bz2 + sha256: 56ecf5446fb6db334e5dca1b5a7cfd91cceb15090cec7dbaab6c31cf4c65365e + md5: 0da43a7c1e52e9b864035831636d2876 + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0a0 + - bioconductor-biocparallel >=1.40.0,<1.41.0 + - bioconductor-biocparallel >=1.40.0,<1.41.0a0 + - bioconductor-biostrings >=2.74.0,<2.75.0 + - bioconductor-biostrings >=2.74.0,<2.75.0a0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0a0 + - bioconductor-genomicranges >=1.58.0,<1.59.0 + - bioconductor-genomicranges >=1.58.0,<1.59.0a0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-iranges >=2.40.0,<2.41.0a0 + - bioconductor-rsamtools >=2.22.0,<2.23.0 + - bioconductor-rsamtools >=2.22.0,<2.23.0a0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0a0 + - bioconductor-summarizedexperiment >=1.36.0,<1.37.0 + - bioconductor-summarizedexperiment >=1.36.0,<1.37.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + size: 2458572 + timestamp: 1744676929879 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-genomicfeatures-1.58.0-r44hdfd78af_0.tar.bz2 + sha256: a505f548b425830b61e0edb326235140b9f30b097e6312d8dd6c8882814495c1 + md5: b48b0aac0547a72e987f92a80cbffaf2 + depends: + - bioconductor-annotationdbi >=1.68.0,<1.69.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biostrings >=2.74.0,<2.75.0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0 + - bioconductor-genomicranges >=1.58.0,<1.59.0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-rtracklayer >=1.66.0,<1.67.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-xvector >=0.46.0,<0.47.0 + - r-base >=4.4,<4.5.0a0 + - r-dbi + license: Artistic-2.0 + size: 1471012 + timestamp: 1735052808971 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-genomicranges-1.58.0-r44h3df3fcb_2.tar.bz2 + sha256: 175012b124573b042d9733230f42ddc1b40a27de04e70b03703159957164de99 + md5: 2200bf0109f17b793a1e0140f74e0d1a + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0a0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0a0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-iranges >=2.40.0,<2.41.0a0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0a0 + - bioconductor-xvector >=0.46.0,<0.47.0 + - bioconductor-xvector >=0.46.0,<0.47.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + size: 2552945 + timestamp: 1738827563344 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-go.db-3.20.0-r44hdfd78af_0.tar.bz2 + sha256: 603de2a2290d00c04b42fc3e94727da7cc2ea844c32d6d6d2a330de08eb31d64 + md5: cb2a625288943786368e8423d925ffad + depends: + - bioconductor-annotationdbi >=1.68.0,<1.69.0 + - bioconductor-data-packages >=20241103 + - curl + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + size: 9006 + timestamp: 1734919500598 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-goseq-1.58.0-r44hdfd78af_0.tar.bz2 + sha256: 37b094acb8b36ad7629e4fa60670ab1d297539e9f65c79beeb3c32f91f9253a6 + md5: 83a3b3a873c5bebd186e4b48727adea2 + depends: + - bioconductor-annotationdbi >=1.68.0,<1.69.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-genelendatabase >=1.42.0,<1.43.0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0 + - bioconductor-genomicfeatures >=1.58.0,<1.59.0 + - bioconductor-go.db >=3.20.0,<3.21.0 + - bioconductor-rtracklayer >=1.66.0,<1.67.0 + - r-base >=4.4,<4.5.0a0 + - r-biasedurn + - r-mgcv + license: LGPL (>= 2) + size: 1938603 + timestamp: 1735069393180 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-iranges-2.40.0-r44h3df3fcb_2.tar.bz2 + sha256: 5103487434961fb2c62af257d45663200fe2cbbd50b861678d95e0ec46556648 + md5: f2e8f02a2987ccaf39a22c36fa0e9fa8 + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0a0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + size: 2618729 + timestamp: 1738085553927 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-keggrest-1.46.0-r44hdfd78af_0.tar.bz2 + sha256: d16c8d24cf2f04e3392081111cd8b8690c0e33ebea77d2208c236a7a0e46645c + md5: dd1a6a8509dc43cec9a22754eec1bef3 + depends: + - bioconductor-biostrings >=2.74.0,<2.75.0 + - r-base >=4.4,<4.5.0a0 + - r-httr + - r-png + license: Artistic-2.0 + size: 435387 + timestamp: 1734646496598 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-limma-3.62.1-r44h15a9599_1.tar.bz2 + sha256: 1aa970c6ca0ffa1a412b19c1a09c19a6061f79db95a9bf424c47bd8e4cc9a645 + md5: f98d28949c8fbfcfa8e7073512d4bbee + depends: + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - liblzma >=5.6.4,<6.0a0 + - libzlib >=1.3.1,<2.0a0 + - r-base >=4.4,<4.5.0a0 + - r-statmod + license: GPL (>=2) + size: 3209741 + timestamp: 1738707632241 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-matrixgenerics-1.18.0-r44hdfd78af_0.tar.bz2 + sha256: 4285ed22931da41e368946c3d89901df479421a148ef83ad2c0f1378c5f59a28 + md5: d1b86fcb6d7e4d3c9fe67817c739b5a7 + depends: + - r-base >=4.4,<4.5.0a0 + - r-matrixstats >=1.4.1 + license: Artistic-2.0 + size: 505327 + timestamp: 1734198685067 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-qvalue-2.38.0-r44hdfd78af_0.tar.bz2 + sha256: 41264a1aedf63b1d335af56828a7d42f41bbbbac68e107d4e5bfbddfd066cc6e + md5: 68d6f0e442750f3aab1d71b8f8c4e0bb + depends: + - r-base >=4.4,<4.5.0a0 + - r-ggplot2 + - r-reshape2 + license: LGPL + size: 2849292 + timestamp: 1734185009029 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-rhtslib-3.2.0-r44h15a9599_2.tar.bz2 + sha256: b0dc13067a74531d912e31e2dd50e00214642c0a87dbe94c0f0841ca2151be5c + md5: cd4f1045bdf924b91a2cbb32e6af1fc3 + depends: + - bioconductor-zlibbioc >=1.52.0,<1.53.0 + - bioconductor-zlibbioc >=1.52.0,<1.53.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - liblzma >=5.8.1,<6.0a0 + - libzlib >=1.3.1,<2.0a0 + - r-base >=4.4,<4.5.0a0 + license: LGPL (>= 2) + size: 2562091 + timestamp: 1744651787460 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-rsamtools-2.22.0-r44h77050f0_1.tar.bz2 + sha256: 05456081875d82b154b28052b33577f10fbd3965c99f69fd6f79911ba700f049 + md5: 954a9309fd39dfa2b398b4a1770dbb9a + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0a0 + - bioconductor-biocparallel >=1.40.0,<1.41.0 + - bioconductor-biocparallel >=1.40.0,<1.41.0a0 + - bioconductor-biostrings >=2.74.0,<2.75.0 + - bioconductor-biostrings >=2.74.0,<2.75.0a0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0a0 + - bioconductor-genomicranges >=1.58.0,<1.59.0 + - bioconductor-genomicranges >=1.58.0,<1.59.0a0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-iranges >=2.40.0,<2.41.0a0 + - bioconductor-rhtslib >=3.2.0,<3.3.0 + - bioconductor-rhtslib >=3.2.0,<3.3.0a0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0a0 + - bioconductor-xvector >=0.46.0,<0.47.0 + - bioconductor-xvector >=0.46.0,<0.47.0a0 + - bioconductor-zlibbioc >=1.52.0,<1.53.0 + - bioconductor-zlibbioc >=1.52.0,<1.53.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - liblzma >=5.8.1,<6.0a0 + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + - r-base >=4.4,<4.5.0a0 + - r-bitops + license: Artistic-2.0 | file LICENSE + size: 4270321 + timestamp: 1744652153795 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-rtracklayer-1.66.0-r44h15a9599_1.tar.bz2 + sha256: 5ce937215c5f3aef4c0dcbc63db606fb406796d75f03a186a33fe6f0e2fe5693 + md5: 722f9e9393267173248af400a48c6ac2 + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0a0 + - bioconductor-biocio >=1.16.0,<1.17.0 + - bioconductor-biocio >=1.16.0,<1.17.0a0 + - bioconductor-biostrings >=2.74.0,<2.75.0 + - bioconductor-biostrings >=2.74.0,<2.75.0a0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0a0 + - bioconductor-genomicalignments >=1.42.0,<1.43.0 + - bioconductor-genomicalignments >=1.42.0,<1.43.0a0 + - bioconductor-genomicranges >=1.58.0,<1.59.0 + - bioconductor-genomicranges >=1.58.0,<1.59.0a0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-iranges >=2.40.0,<2.41.0a0 + - bioconductor-rsamtools >=2.22.0,<2.23.0 + - bioconductor-rsamtools >=2.22.0,<2.23.0a0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0a0 + - bioconductor-xvector >=0.46.0,<0.47.0 + - bioconductor-xvector >=0.46.0,<0.47.0a0 + - bioconductor-zlibbioc >=1.52.0,<1.53.0 + - bioconductor-zlibbioc >=1.52.0,<1.53.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - libzlib >=1.3.1,<2.0a0 + - r-base >=4.4,<4.5.0a0 + - r-curl + - r-httr + - r-restfulr >=0.0.13 + - r-restfulr >=0.0.15,<0.1.0a0 + - r-xml >=1.98-0 + license: Artistic-2.0 + file LICENSE + size: 5622285 + timestamp: 1744677326188 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-s4arrays-1.6.0-r44h3df3fcb_1.tar.bz2 + sha256: 8ec7155203462c542da0565c702261a87ba5cf7b5b2a6998fdbc9eca728d453c + md5: a6774527b21da1eb7a99b3839301ab57 + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0a0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-iranges >=2.40.0,<2.41.0a0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - r-abind + - r-base >=4.4,<4.5.0a0 + - r-crayon + - r-matrix + license: Artistic-2.0 + size: 1069423 + timestamp: 1744662789860 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-s4vectors-0.44.0-r44h3df3fcb_2.tar.bz2 + sha256: 08b6d33a61aca360a98d2188bc4ca9d2a780683e280c4b789cdb7d285719e07e + md5: 13bdbde9c9496802b7974d006f22fe11 + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + size: 2780446 + timestamp: 1738085295539 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-seqlogo-1.72.0-r44hdfd78af_0.tar.bz2 + sha256: 57c6528aa7d0cc525c33fe8fcfdc98f4fed1073a3a97569fd478283ba4929ede + md5: 0e5767fd35b1907a6f26a36adf678927 + depends: + - r-base >=4.4,<4.5.0a0 + license: LGPL (>= 2) + size: 645174 + timestamp: 1734197222701 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-sparsearray-1.6.0-r44h3df3fcb_1.tar.bz2 + sha256: 244993331f46f90037905c31f17e2aa49871d366f2b0f40817942a15c3a2426f + md5: 143242d9cf4199b8f26c2552b61d2049 + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0a0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-iranges >=2.40.0,<2.41.0a0 + - bioconductor-matrixgenerics >=1.18.0,<1.19.0 + - bioconductor-matrixgenerics >=1.18.0,<1.19.0a0 + - bioconductor-s4arrays >=1.6.0,<1.7.0 + - bioconductor-s4arrays >=1.6.0,<1.7.0a0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0a0 + - bioconductor-xvector >=0.46.0,<0.47.0 + - bioconductor-xvector >=0.46.0,<0.47.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - r-base >=4.4,<4.5.0a0 + - r-matrix + - r-matrixstats + license: Artistic-2.0 + size: 1878617 + timestamp: 1744663082341 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-summarizedexperiment-1.36.0-r44hdfd78af_0.tar.bz2 + sha256: 163f500839b6aee3d5b1b836ffe2241d84188cf2047a15f6333d4b54cb8d3157 + md5: 144e795fdb25d213899e249bcb538bc4 + depends: + - bioconductor-biobase >=2.66.0,<2.67.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-delayedarray >=0.32.0,<0.33.0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0 + - bioconductor-genomicranges >=1.58.0,<1.59.0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-matrixgenerics >=1.18.0,<1.19.0 + - bioconductor-s4arrays >=1.6.0,<1.7.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - r-base >=4.4,<4.5.0a0 + - r-matrix + license: Artistic-2.0 + size: 1910957 + timestamp: 1734708289157 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/bioconductor-txdbmaker-1.2.0-r44hdfd78af_0.tar.bz2 + sha256: 07288c7c4fdbc193145c671af603e6278c78d74ee0a0aa493f00a229a280caa8 + md5: d25b525f31267207fff3f05ed1e2c859 + depends: + - bioconductor-annotationdbi >=1.68.0,<1.69.0 + - bioconductor-biobase >=2.66.0,<2.67.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocio >=1.16.0,<1.17.0 + - bioconductor-biomart >=2.62.0,<2.63.0 + - bioconductor-genomeinfodb >=1.42.0,<1.43.0 + - bioconductor-genomicfeatures >=1.58.0,<1.59.0 + - bioconductor-genomicranges >=1.58.0,<1.59.0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-rtracklayer >=1.66.0,<1.67.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-ucsc.utils >=1.2.0,<1.3.0 + - r-base >=4.4,<4.5.0a0 + - r-dbi + - r-httr + - r-rjson + - r-rsqlite >=2.0 + license: Artistic-2.0 + size: 1251769 + timestamp: 1735056720217 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-ucsc.utils-1.2.0-r44h9ee0642_1.tar.bz2 + sha256: b44c061b88b91fd7666d483f1fa7de46532fb936c74e54e30d31900181f3d18d + md5: 1653b8e9eb24949cc045e21ad21927b6 + depends: + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0a0 + - r-base >=4.4,<4.5.0a0 + - r-httr + - r-jsonlite + license: Artistic-2.0 + size: 315595 + timestamp: 1738085766116 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-xvector-0.46.0-r44h15a9599_2.tar.bz2 + sha256: 0e9ff1ff36d2aef4999ff248785e7bde5ffc825cd95466a1df52732d45cf9c77 + md5: 1ce6efc6cb59e8ce10ec4cb722ece013 + depends: + - bioconductor-biocgenerics >=0.52.0,<0.53.0 + - bioconductor-biocgenerics >=0.52.0,<0.53.0a0 + - bioconductor-iranges >=2.40.0,<2.41.0 + - bioconductor-iranges >=2.40.0,<2.41.0a0 + - bioconductor-s4vectors >=0.44.0,<0.45.0 + - bioconductor-s4vectors >=0.44.0,<0.45.0a0 + - bioconductor-zlibbioc >=1.52.0,<1.53.0 + - bioconductor-zlibbioc >=1.52.0,<1.53.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - libzlib >=1.3.1,<2.0a0 + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + size: 743866 + timestamp: 1738086002632 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bioconductor-zlibbioc-1.52.0-r44h3df3fcb_2.tar.bz2 + sha256: 9d9c1f20cec2bac9c259b1e7798795e61c92573cfea74b5fda739c37a118cf6a + md5: 1f08126acbb2d9a8835aba0b1a7bf9e2 + depends: + - libblas >=3.9.0,<4.0a0 + - libgcc >=13 + - liblapack >=3.9.0,<4.0a0 + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + file LICENSE + size: 254875 + timestamp: 1738085054154 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/biopython-1.86-py311h49ec1c0_0.conda + sha256: edb6339d4714f642d7d97649b4e379cc513a15f1e666a4a3d19fec4a883c2296 + md5: e52878a5c94017b6f7c033ba7d26bc6b + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - numpy + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: LicenseRef-Biopython + size: 3323306 + timestamp: 1761734827058 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/blast-2.17.0-h66d330f_0.conda + sha256: 53a319c984d8aafcf7f2d7d4c21ad2590e1bacd674443e09d79ac19694ccf99a + md5: 405ce6d52eba06fcd48197ae1eb8f5a9 + depends: + - bzip2 >=1.0.8,<2.0a0 + - curl + - entrez-direct >=24.0,<25.0a0 + - libgcc >=13 + - libsqlite >=3.50.4,<4.0a0 + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + - ncbi-vdb >=3.2.1,<4.0a0 + - perl + - perl-archive-tar + - perl-json + - perl-list-moreutils + - zlib + license: NCBI-PD + size: 84832339 + timestamp: 1754909742570 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/boltons-25.0.0-pyhd8ed1ab_0.conda + sha256: ea5f4c876eff2ed469551b57f1cc889a3c01128bf3e2e10b1fea11c3ef39eac2 + md5: c7eb87af73750d6fd97eff8bbee8cb9c + depends: + - python >=3.9 + license: BSD-3-Clause + license_family: BSD + size: 302296 + timestamp: 1749686302834 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/boost-cpp-1.85.0-h3c6214e_4.conda + sha256: dec1f52d8869b94a9cb2b4d9a727a00ae2c54e7b5a66a51d3d9f1c33159ebb45 + md5: cc4533eabf5caa8b4fbb56418d1617a9 + depends: + - bzip2 >=1.0.8,<2.0a0 + - icu >=75.1,<76.0a0 + - libboost-devel 1.85.0 h00ab1b0_4 + - libzlib >=1.3.1,<2.0a0 + - xz >=5.2.6,<6.0a0 + - zstd >=1.5.6,<1.6.0a0 + license: BSL-1.0 + size: 18122 + timestamp: 1722289881184 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/bowtie2-2.5.4-he96a11b_6.tar.bz2 + sha256: cab9a873afdce717d833003b06dc42e20773563c5c186cfb2ae2fe0613c83bbf + md5: 17f2305502aa2d28a3747db4ad62a28d + depends: + - _openmp_mutex >=4.5 + - libgcc >=13 + - libgomp + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + - perl + - python + - zstd >=1.5.7,<1.6.0a0 + license: GPL-3.0-or-later + license_family: GPL3 + size: 14983732 + timestamp: 1748989756904 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/brotli-1.2.0-h41a2e66_0.conda + sha256: 33239a07f7685917cac25646dd33798ee93e61f83504a0c938d86c507e05d7c9 + md5: 4ddfd44e473c676cb8e80548ba4aa704 + depends: + - __glibc >=2.17,<3.0.a0 + - brotli-bin 1.2.0 hf2c8021_0 + - libbrotlidec 1.2.0 hd53d788_0 + - libbrotlienc 1.2.0 h02bd7ab_0 + - libgcc >=14 + license: MIT + license_family: MIT + size: 19964 + timestamp: 1761592234411 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/brotli-bin-1.2.0-hf2c8021_0.conda + sha256: b4aa87fa7658c79e9334c607ad399a964ff75ec8241b9b744b8dc8fc84b55dd0 + md5: 5304333319a6124a2737d9f128cbc4ed + depends: + - __glibc >=2.17,<3.0.a0 + - libbrotlidec 1.2.0 hd53d788_0 + - libbrotlienc 1.2.0 h02bd7ab_0 + - libgcc >=14 + license: MIT + license_family: MIT + size: 20993 + timestamp: 1761592224816 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/brotli-python-1.2.0-py311h7c6b74e_0.conda + sha256: 5e6858dae1935793a7fa7f46d8975b0596b546c28586cb463dd2fdeba3bcc193 + md5: 645bc783bc723d67a294a51bc860762d + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + constrains: + - libbrotlicommon 1.2.0 h09219d5_0 + license: MIT + license_family: MIT + size: 368532 + timestamp: 1761592301216 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/busco-6.0.0-pyhdfd78af_1.conda + sha256: 3b71b256468403e5e2984884f6db7964c8155e1627888eddb7e37631d5e88d23 + md5: 5c4fcbf8312c745e5186232a57b67268 + depends: + - augustus >=3.3 + - bbmap + - biopython >=1.79 + - blast >=2.10.1 + - fonts-conda-ecosystem + - hmmer >=3.1b2 + - matplotlib-base + - metaeuk >=6.a5d39d9 + - miniprot + - pandas + - prodigal + - python >=3.3 + - requests + - sepp >=4.5.6 + - wget + license: MIT + license_family: MIT + size: 348950 + timestamp: 1759927722737 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/bwidget-1.10.1-ha770c72_1.conda + sha256: c88dd33c89b33409ebcd558d78fdc66a63c18f8b06e04d170668ffb6c8ecfabd + md5: 983b92277d78c0d0ec498e460caa0e6d + depends: + - tk + license: TCL + size: 129594 + timestamp: 1750261567920 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/bzip2-1.0.8-hda65f42_8.conda + sha256: c30daba32ddebbb7ded490f0e371eae90f51e72db620554089103b4a6934b0d5 + md5: 51a19bba1b8ebfb60df25cde030b7ebc + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + license: bzip2-1.0.6 + license_family: BSD + size: 260341 + timestamp: 1757437258798 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/c-ares-1.34.5-hb9d3cd8_0.conda + sha256: f8003bef369f57396593ccd03d08a8e21966157269426f71e943f96e4b579aeb + md5: f7f0d6cc2dc986d42ac2689ec88192be + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: MIT + license_family: MIT + size: 206884 + timestamp: 1744127994291 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/ca-certificates-2025.11.12-hbd8a1cb_0.conda + sha256: b986ba796d42c9d3265602bc038f6f5264095702dd546c14bc684e60c385e773 + md5: f0991f0f84902f6b6009b4d2350a83aa + depends: + - __unix + license: ISC + size: 152432 + timestamp: 1762967197890 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/cairo-1.18.4-h3394656_0.conda + sha256: 3bd6a391ad60e471de76c0e9db34986c4b5058587fbf2efa5a7f54645e28c2c7 + md5: 09262e66b19567aff4f592fb53b28760 + depends: + - __glibc >=2.17,<3.0.a0 + - fontconfig >=2.15.0,<3.0a0 + - fonts-conda-ecosystem + - freetype >=2.12.1,<3.0a0 + - icu >=75.1,<76.0a0 + - libexpat >=2.6.4,<3.0a0 + - libgcc >=13 + - libglib >=2.82.2,<3.0a0 + - libpng >=1.6.47,<1.7.0a0 + - libstdcxx >=13 + - libxcb >=1.17.0,<2.0a0 + - libzlib >=1.3.1,<2.0a0 + - pixman >=0.44.2,<1.0a0 + - xorg-libice >=1.1.2,<2.0a0 + - xorg-libsm >=1.2.5,<2.0a0 + - xorg-libx11 >=1.8.11,<2.0a0 + - xorg-libxext >=1.3.6,<2.0a0 + - xorg-libxrender >=0.9.12,<0.10.0a0 + license: LGPL-2.1-only or MPL-1.1 + size: 978114 + timestamp: 1741554591855 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/cd-hit-4.8.1-h5ca1c30_13.tar.bz2 + sha256: 8793a98a7c93b9ade5caaa85e6eeccec69d774c197b772a8132373de2d6ce8e1 + md5: 8b5beac305bcf38b77be27e0233e4076 + depends: + - _openmp_mutex >=4.5 + - libgcc >=13 + - libgomp + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + license: GPL-2.0-or-later + license_family: GPL + size: 185148 + timestamp: 1745532039886 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/cdbtools-0.99-h077b44d_12.tar.bz2 + sha256: b70324c9af58110d40b8ed9b5210fe7829cadd59ffef66cf01fa818cc9476fa6 + md5: 6b938da47db28f06ca9258f92d80e7cb + depends: + - libgcc >=13 + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + license: Public Domain + size: 80142 + timestamp: 1744771014271 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/certifi-2025.11.12-pyhd8ed1ab_0.conda + sha256: 083a2bdad892ccf02b352ecab38ee86c3e610ba9a4b11b073ea769d55a115d32 + md5: 96a02a5c1a65470a7e4eedb644c872fd + depends: + - python >=3.10 + license: ISC + size: 157131 + timestamp: 1762976260320 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/cffi-2.0.0-py311h03d9500_1.conda + sha256: 3ad13377356c86d3a945ae30e9b8c8734300925ef81a3cb0a9db0d755afbe7bb + md5: 3912e4373de46adafd8f1e97e4bd166b + depends: + - __glibc >=2.17,<3.0.a0 + - libffi >=3.5.2,<3.6.0a0 + - libgcc >=14 + - pycparser + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: MIT + license_family: MIT + size: 303338 + timestamp: 1761202960110 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/charset-normalizer-3.4.4-pyhd8ed1ab_0.conda + sha256: b32f8362e885f1b8417bac2b3da4db7323faa12d5db62b7fd6691c02d60d6f59 + md5: a22d1fd9bf98827e280a02875d9a007a + depends: + - python >=3.10 + license: MIT + license_family: MIT + size: 50965 + timestamp: 1760437331772 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/clustalw-2.1-h9948957_12.tar.bz2 + sha256: d82a95c35cfdb590c266131fdbb0ec62c7977a1edbc95ac43a03220886c855a4 + md5: 5989fcb871dabc90d1bb5de3d5c79185 + depends: + - libgcc >=13 + - libstdcxx >=13 + license: LGPL-3.0-or-later + license_family: LGPL + size: 384703 + timestamp: 1738739696573 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/colorama-0.4.6-pyhd8ed1ab_1.conda + sha256: ab29d57dc70786c1269633ba3dff20288b81664d3ff8d21af995742e2bb03287 + md5: 962b9857ee8e7018c22f2776ffa0b2d7 + depends: + - python >=3.9 + license: BSD-3-Clause + license_family: BSD + size: 27011 + timestamp: 1733218222191 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/conda-25.9.1-py311h38be061_0.conda + sha256: 3afe2757f8d4885dca575fd33a41521548955ba18ae77199dd6103787e5766b6 + md5: 6d2e5856a4a3d51370c750938e9d502a + depends: + - archspec >=0.2.3 + - boltons >=23.0.0 + - charset-normalizer + - conda-libmamba-solver >=25.4.0 + - conda-package-handling >=2.2.0 + - distro >=1.5.0 + - frozendict >=2.4.2 + - jsonpatch >=1.32 + - menuinst >=2 + - packaging >=23.0 + - platformdirs >=3.10.0 + - pluggy >=1.0.0 + - pycosat >=0.6.3 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + - requests >=2.28.0,<3 + - ruamel.yaml >=0.11.14,<0.19 + - setuptools >=60.0.0 + - tqdm >=4 + - truststore >=0.8.0 + - zstandard >=0.19.0 + constrains: + - conda-build >=25.9 + - conda-content-trust >=0.1.1 + - conda-env >=2.6 + license: BSD-3-Clause + license_family: BSD + size: 1235353 + timestamp: 1760108681826 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/conda-libmamba-solver-25.4.0-pyhd8ed1ab_0.conda + sha256: 48999a7a6e300075e4ef1c85130614d75429379eea8fe78f18a38a8aab8da384 + md5: d62b8f745ff471d5594ad73605cb9b59 + depends: + - boltons >=23.0.0 + - conda >=24.11 + - libmambapy >=2.0.0 + - python >=3.9 + license: BSD-3-Clause + license_family: BSD + size: 41985 + timestamp: 1745834587643 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/conda-package-handling-2.4.0-pyh7900ff3_2.conda + sha256: 8b2b1c235b7cbfa8488ad88ff934bdad25bac6a4c035714681fbff85b602f3f0 + md5: 32c158f481b4fd7630c565030f7bc482 + depends: + - conda-package-streaming >=0.9.0 + - python >=3.9 + - requests + - zstandard >=0.15 + license: BSD-3-Clause + license_family: BSD + size: 257995 + timestamp: 1736345601691 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/conda-package-streaming-0.12.0-pyhd8ed1ab_0.conda + sha256: 11b76b0be2f629e8035be1d723ccb6e583eb0d2af93bde56113da7fa6e2f2649 + md5: ff75d06af779966a5aeae1be1d409b96 + depends: + - python >=3.9 + - zstandard >=0.15 + license: BSD-3-Clause + license_family: BSD + size: 21933 + timestamp: 1751548225624 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/contourpy-1.3.3-py311hdf67eae_3.conda + sha256: fde69b5ab61225daca6c2f05a93f94c06af93003e4f871d61470df5c4cf9587b + md5: c4e2f4d5193e55a70bb67a2aa07006ae + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - numpy >=1.25 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: BSD-3-Clause + size: 296142 + timestamp: 1762525422359 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/coreutils-9.5-hd590300_0.conda + sha256: 7cd3b0f55aa55bb27b045c30f32b3f6b874ecc006f3abcb274c71a3bcbacb358 + md5: 126d457e0e7a535278e808a7d8960015 + depends: + - libgcc-ng >=12 + license: GPL-3.0-or-later + license_family: GPL + size: 3014238 + timestamp: 1711655132451 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/cpp-expected-1.3.1-h171cf75_0.conda + sha256: 0d9405d9f2de5d4b15d746609d87807aac10e269072d6408b769159762ed113d + md5: d17488e343e4c5c0bd0db18b3934d517 + depends: + - libstdcxx >=14 + - libgcc >=14 + - __glibc >=2.17,<3.0.a0 + license: CC0-1.0 + size: 24283 + timestamp: 1756734785482 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/curl-8.17.0-h4e3cde8_0.conda + sha256: 3fb39c401fbdbaf68b8f25c1d81600d2a771b6467cc5d7c88fbd1e06d8825ee1 + md5: a37bd62e2c34797cdb577920b35f3bc5 + depends: + - __glibc >=2.17,<3.0.a0 + - krb5 >=1.21.3,<1.22.0a0 + - libcurl 8.17.0 h4e3cde8_0 + - libgcc >=14 + - libssh2 >=1.11.1,<2.0a0 + - libzlib >=1.3.1,<2.0a0 + - openssl >=3.5.4,<4.0a0 + - zstd >=1.5.7,<1.6.0a0 + license: curl + license_family: MIT + size: 186150 + timestamp: 1762333752178 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/cycler-0.12.1-pyhd8ed1ab_1.conda + sha256: 9827efa891e507a91a8a2acf64e210d2aff394e1cde432ad08e1f8c66b12293c + md5: 44600c4667a319d67dbe0681fc0bc833 + depends: + - python >=3.9 + license: BSD-3-Clause + license_family: BSD + size: 13399 + timestamp: 1733332563512 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/cyrus-sasl-2.1.28-hd9c7081_0.conda + sha256: ee09ad7610c12c7008262d713416d0b58bf365bc38584dce48950025850bdf3f + md5: cae723309a49399d2949362f4ab5c9e4 + depends: + - __glibc >=2.17,<3.0.a0 + - krb5 >=1.21.3,<1.22.0a0 + - libgcc >=13 + - libntlm >=1.8,<2.0a0 + - libstdcxx >=13 + - libxcrypt >=4.4.36 + - openssl >=3.5.0,<4.0a0 + license: BSD-3-Clause-Attribution + license_family: BSD + size: 209774 + timestamp: 1750239039316 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/dbus-1.16.2-h3c4dab8_0.conda + sha256: 3b988146a50e165f0fa4e839545c679af88e4782ec284cc7b6d07dd226d6a068 + md5: 679616eb5ad4e521c83da4650860aba7 + depends: + - libstdcxx >=13 + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libexpat >=2.7.0,<3.0a0 + - libzlib >=1.3.1,<2.0a0 + - libglib >=2.84.2,<3.0a0 + license: GPL-2.0-or-later + license_family: GPL + size: 437860 + timestamp: 1747855126005 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/dendropy-5.0.8-pyhdfd78af_1.tar.bz2 + sha256: 102790b465893f5a5c4c3fa045f677e23f990ecc6d1b61d240153c3569a5b99f + md5: 0fa94922431d1074792639704d19461d + depends: + - python >=3.6 + license: BSD-3-Clause + license_family: BSD + size: 335113 + timestamp: 1750372782364 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/diamond-2.1.16-h13889ed_0.conda + sha256: ad1ec8a459744d513fc67c99685b9b1f559635a7cc5d8392acc02311f2f9e3b4 + md5: 778c9cf6515aa20b494f4954f60f2983 + depends: + - _openmp_mutex >=4.5 + - libgcc >=13 + - libgomp + - libsqlite >=3.51.0,<4.0a0 + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + license: GPL-3.0-or-later + license_family: GPL3 + size: 16884344 + timestamp: 1762807031736 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/distro-1.9.0-pyhd8ed1ab_1.conda + sha256: 5603c7d0321963bb9b4030eadabc3fd7ca6103a38475b4e0ed13ed6d97c86f4e + md5: 0a2014fd9860f8b1eaa0b1f3d3771a08 + depends: + - python >=3.9 + license: Apache-2.0 + license_family: APACHE + size: 41773 + timestamp: 1734729953882 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/dnspython-2.8.0-pyhcf101f3_0.conda + sha256: ef1e7b8405997ed3d6e2b6722bd7088d4a8adf215e7c88335582e65651fb4e05 + md5: d73fdc05f10693b518f52c994d748c19 + depends: + - python >=3.10,<4.0.0 + - sniffio + - python + constrains: + - aioquic >=1.2.0 + - cryptography >=45 + - httpcore >=1.0.0 + - httpx >=0.28.0 + - h2 >=4.2.0 + - idna >=3.10 + - trio >=0.30 + - wmi >=1.5.1 + license: ISC + size: 196500 + timestamp: 1757292856922 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/double-conversion-3.3.1-h5888daf_0.conda + sha256: 1bcc132fbcc13f9ad69da7aa87f60ea41de7ed4d09f3a00ff6e0e70e1c690bc2 + md5: bfd56492d8346d669010eccafe0ba058 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libstdcxx >=13 + license: BSD-3-Clause + license_family: BSD + size: 69544 + timestamp: 1739569648873 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/entrez-direct-24.0-he881be0_0.tar.bz2 + sha256: 71a8f349659c9c18efa544663de2db1a20b5e3d32f8e4e88cd33110a0caf4eb3 + md5: 52a3fabee9201c2c6093c13b4eaf29b4 + depends: + - wget + license: Public Domain + size: 17127014 + timestamp: 1748473197798 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/ete3-3.1.3-pyhd8ed1ab_0.conda + sha256: 003a3f1939e89cc2e4139882030d9c329cee92aee6299d69a8d2fa671b576fb3 + md5: 7fca9be27a33c9244676f43cc0bb792e + depends: + - lxml + - numpy + - pyqt >=5.15.* + - python >=3.7 + - scipy + - six + - xorg-libsm + - xorg-libxau + - xorg-libxdmcp + - xorg-libxext + - xorg-libxrender + - xorg-xextproto + license: GPL-3.0-only + license_family: GPL + size: 1778742 + timestamp: 1688510187352 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/famsa-2.4.1-h9ee0642_0.tar.bz2 + sha256: c4a54b4199ffa7e63b162dc3514889d3481531cdd9fd41b86e83bc049630cd93 + md5: 315177feab042829efe7a14054b5ac53 + license: GPL-3.0-or-later + license_family: GPL3 + size: 1389689 + timestamp: 1752619191048 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/fastme-2.1.6.3-h7b50bb2_1.tar.bz2 + sha256: a99eb1eb5bf1e6db0cab86423c404eb3b0922b6d36b5511ba77597c4e1346761 + md5: 4c2c2ac65080caab30d3b735eb867388 + depends: + - libgcc >=13 + license: GPL-3.0-only + license_family: GPL3 + size: 152311 + timestamp: 1733934328397 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/fastp-1.0.1-heae3180_0.tar.bz2 + sha256: accee7e586711b20b32a47e85fd9ede4c428cd588c0ad5df0be84077f689af66 + md5: 166517452931b0bd1e04fbb06d5e7915 + depends: + - isa-l >=2.31.1,<3.0a0 + - libdeflate >=1.22,<1.26.0a0 + - libgcc >=13 + - libstdcxx >=13 + license: MIT + license_family: MIT + size: 270543 + timestamp: 1750159498639 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/fasttree-2.2.0-h7b50bb2_0.conda + sha256: 6ad495bb05b10cccc909af074360ad275abdce3e8d98c7738e9cf5757f217e79 + md5: 7e98e9bf25eca335b0d81d9e8905e916 + depends: + - _openmp_mutex >=4.5 + - libgcc >=13 + - libgomp + license: GPL-2.0-or-later + license_family: GPL + size: 204873 + timestamp: 1756944351590 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/fmt-11.2.0-h07f6e7f_0.conda + sha256: e0f53b7801d0bcb5d61a1ddcb873479bfe8365e56fd3722a232fbcc372a9ac52 + md5: 0c2f855a88fab6afa92a7aa41217dc8e + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libstdcxx >=13 + license: MIT + license_family: MIT + size: 192721 + timestamp: 1751277120358 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/font-ttf-dejavu-sans-mono-2.37-hab24e00_0.tar.bz2 + sha256: 58d7f40d2940dd0a8aa28651239adbf5613254df0f75789919c4e6762054403b + md5: 0c96522c6bdaed4b1566d11387caaf45 + license: BSD-3-Clause + license_family: BSD + size: 397370 + timestamp: 1566932522327 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/font-ttf-inconsolata-3.000-h77eed37_0.tar.bz2 + sha256: c52a29fdac682c20d252facc50f01e7c2e7ceac52aa9817aaf0bb83f7559ec5c + md5: 34893075a5c9e55cdafac56607368fc6 + license: OFL-1.1 + license_family: Other + size: 96530 + timestamp: 1620479909603 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/font-ttf-source-code-pro-2.038-h77eed37_0.tar.bz2 + sha256: 00925c8c055a2275614b4d983e1df637245e19058d79fc7dd1a93b8d9fb4b139 + md5: 4d59c254e01d9cde7957100457e2d5fb + license: OFL-1.1 + license_family: Other + size: 700814 + timestamp: 1620479612257 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/font-ttf-ubuntu-0.83-h77eed37_3.conda + sha256: 2821ec1dc454bd8b9a31d0ed22a7ce22422c0aef163c59f49dfdf915d0f0ca14 + md5: 49023d73832ef61042f6a237cb2687e7 + license: LicenseRef-Ubuntu-Font-Licence-Version-1.0 + license_family: Other + size: 1620504 + timestamp: 1727511233259 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/fontconfig-2.15.0-h7e30c49_1.conda + sha256: 7093aa19d6df5ccb6ca50329ef8510c6acb6b0d8001191909397368b65b02113 + md5: 8f5b0b297b59e1ac160ad4beec99dbee + depends: + - __glibc >=2.17,<3.0.a0 + - freetype >=2.12.1,<3.0a0 + - libexpat >=2.6.3,<3.0a0 + - libgcc >=13 + - libuuid >=2.38.1,<3.0a0 + - libzlib >=1.3.1,<2.0a0 + license: MIT + license_family: MIT + size: 265599 + timestamp: 1730283881107 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/fonts-conda-ecosystem-1-0.tar.bz2 + sha256: a997f2f1921bb9c9d76e6fa2f6b408b7fa549edd349a77639c9fe7a23ea93e61 + md5: fee5683a3f04bd15cbd8318b096a27ab + depends: + - fonts-conda-forge + license: BSD-3-Clause + license_family: BSD + size: 3667 + timestamp: 1566974674465 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/fonts-conda-forge-1-hc364b38_1.conda + sha256: 54eea8469786bc2291cc40bca5f46438d3e062a399e8f53f013b6a9f50e98333 + md5: a7970cd949a077b7cb9696379d338681 + depends: + - font-ttf-ubuntu + - font-ttf-inconsolata + - font-ttf-dejavu-sans-mono + - font-ttf-source-code-pro + license: BSD-3-Clause + license_family: BSD + size: 4059 + timestamp: 1762351264405 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/fonttools-4.60.1-py311h3778330_0.conda + sha256: 1c4e796c337faaeb0606bd6291e53e31848921ac78f295f2b671a2dc09f816cb + md5: 91f834f85ac92978cfc3c1c178573e85 + depends: + - __glibc >=2.17,<3.0.a0 + - brotli + - libgcc >=14 + - munkres + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + - unicodedata2 >=15.1.0 + license: MIT + license_family: MIT + size: 2940664 + timestamp: 1759187410840 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/freetype-2.14.1-ha770c72_0.conda + sha256: bf8e4dffe46f7d25dc06f31038cacb01672c47b9f45201f065b0f4d00ab0a83e + md5: 4afc585cd97ba8a23809406cd8a9eda8 + depends: + - libfreetype 2.14.1 ha770c72_0 + - libfreetype6 2.14.1 h73754d4_0 + license: GPL-2.0-only OR FTL + size: 173114 + timestamp: 1757945422243 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/fribidi-1.0.16-hb03c661_0.conda + sha256: 858283ff33d4c033f4971bf440cebff217d5552a5222ba994c49be990dacd40d + md5: f9f81ea472684d75b9dd8d0b328cf655 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + license: LGPL-2.1-or-later + size: 61244 + timestamp: 1757438574066 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/frozendict-2.4.7-py311h49ec1c0_0.conda + sha256: df99ccc63c065875cc150f451aaab03e6733898219cf2f248543fe53b08deff0 + md5: cfc74ea184e02e531732abb17778e9fc + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: LGPL-3.0-only + license_family: LGPL + size: 32010 + timestamp: 1763082979695 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gawk-5.3.1-hcd3d067_0.conda + sha256: ec4ebb9444dccfcbff8a2d19b2811b48a20a58dcd08b29e3851cb930fc0f00d8 + md5: 91d4414ab699180b2b0b10b8112c5a2f + depends: + - __glibc >=2.17,<3.0.a0 + - gmp >=6.3.0,<7.0a0 + - libasprintf >=0.22.5,<1.0a0 + - libgcc >=13 + - libgettextpo >=0.22.5,<1.0a0 + - mpfr >=4.2.1,<5.0a0 + - readline >=8.2,<9.0a0 + license: GPL-3.0-or-later + license_family: GPL + size: 1202471 + timestamp: 1726677363710 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gcc_impl_linux-64-15.2.0-hcacfade_7.conda + sha256: 6a19411e3fe4e4f55509f4b0c374663b3f8903ed5ae1cc94be1b88846c50c269 + md5: 3d75679d5e2bd547cb52b913d73f69ef + depends: + - binutils_impl_linux-64 >=2.40 + - libgcc >=15.2.0 + - libgcc-devel_linux-64 15.2.0 h73f6952_107 + - libgomp >=15.2.0 + - libsanitizer 15.2.0 hb13aed2_7 + - libstdcxx >=15.2.0 + - sysroot_linux-64 + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 77766660 + timestamp: 1759968214246 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gettext-0.25.1-h3f43e3d_1.conda + sha256: cbfa8c80771d1842c2687f6016c5e200b52d4ca8f2cc119f6377f64f899ba4ff + md5: c42356557d7f2e37676e121515417e3b + depends: + - __glibc >=2.17,<3.0.a0 + - gettext-tools 0.25.1 h3f43e3d_1 + - libasprintf 0.25.1 h3f43e3d_1 + - libasprintf-devel 0.25.1 h3f43e3d_1 + - libgcc >=14 + - libgettextpo 0.25.1 h3f43e3d_1 + - libgettextpo-devel 0.25.1 h3f43e3d_1 + - libiconv >=1.18,<2.0a0 + - libstdcxx >=14 + license: LGPL-2.1-or-later AND GPL-3.0-or-later + size: 541357 + timestamp: 1753343006214 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gettext-tools-0.25.1-h3f43e3d_1.conda + sha256: c792729288bdd94f21f25f80802d4c66957b4e00a57f7cb20513f07aadfaff06 + md5: a59c05d22bdcbb4e984bf0c021a2a02f + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libiconv >=1.18,<2.0a0 + license: GPL-3.0-or-later + license_family: GPL + size: 3644103 + timestamp: 1753342966311 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gfortran_impl_linux-64-15.2.0-h1b0a18f_7.conda + sha256: 42e7708bafd41520290e74a904ebdd61049f2de90e8c1e984a889f36a2f5d140 + md5: e3881b0a8332d614c4f44269d655e34a + depends: + - gcc_impl_linux-64 >=15.2.0 + - libgcc >=15.2.0 + - libgfortran5 >=15.2.0 + - libstdcxx >=15.2.0 + - sysroot_linux-64 + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 18408383 + timestamp: 1759968476862 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/giflib-5.2.2-hd590300_0.conda + sha256: aac402a8298f0c0cc528664249170372ef6b37ac39fdc92b40601a6aed1e32ff + md5: 3bf7b9fd5a7136126e0234db4b87c8b6 + depends: + - libgcc-ng >=12 + license: MIT + license_family: MIT + size: 77248 + timestamp: 1712692454246 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/git-2.51.2-pl5321h28be001_0.conda + sha256: 6ec0a715e315bade2408c965d0c25173883ef083a7c432c0c21436d96d7b0c09 + md5: 1e4465f1bb7537c13d299922bbdb6ad7 + depends: + - __glibc >=2.28,<3.0.a0 + - libcurl >=8.16.0,<9.0a0 + - libexpat >=2.7.1,<3.0a0 + - libgcc >=14 + - libiconv >=1.18,<2.0a0 + - libzlib >=1.3.1,<2.0a0 + - openssl >=3.5.4,<4.0a0 + - pcre2 >=10.46,<10.47.0a0 + - perl 5.* + license: GPL-2.0-or-later and LGPL-2.1-or-later + size: 11563641 + timestamp: 1761751922851 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/glib-2.86.1-hbcf1ec1_2.conda + sha256: 437e67015fab762216332c71d5bb91d45f936f91d6c1eb2b77078a12fc979aff + md5: 782f4ffc15f28cfa5dd79bff36a5afd4 + depends: + - glib-tools 2.86.1 hf516916_2 + - libffi >=3.5.2,<3.6.0a0 + - libglib 2.86.1 h32235b2_2 + - packaging + - python * + license: LGPL-2.1-or-later + size: 613149 + timestamp: 1762787549223 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/glib-tools-2.86.1-hf516916_2.conda + sha256: 743c57390c289c771a3bc90e27c817322a6dc518a3f00970caf2ee7b09421b46 + md5: b069da7bb5db4edd45e9f8887f10b52e + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libglib 2.86.1 h32235b2_2 + license: LGPL-2.1-or-later + size: 115896 + timestamp: 1762787507126 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/glpk-5.0-h445213a_0.tar.bz2 + sha256: 0e19c61198ae9e188c43064414a40101f5df09970d4a2c483c0c46a6b1538966 + md5: efc4b0c33bdf47312ad5a8a0587fa653 + depends: + - gmp >=6.2.1,<7.0a0 + - libgcc-ng >=9.3.0 + license: GPL-3.0-or-later + license_family: GPL + size: 1047292 + timestamp: 1624569176979 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gmp-6.3.0-hac33072_2.conda + sha256: 309cf4f04fec0c31b6771a5809a1909b4b3154a2208f52351e1ada006f4c750c + md5: c94a5994ef49749880a8139cf9afcbe1 + depends: + - libgcc-ng >=12 + - libstdcxx-ng >=12 + license: GPL-2.0-or-later OR LGPL-3.0-or-later + size: 460055 + timestamp: 1718980856608 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/graphite2-1.3.14-hecca717_2.conda + sha256: 25ba37da5c39697a77fce2c9a15e48cf0a84f1464ad2aafbe53d8357a9f6cc8c + md5: 2cd94587f3a401ae05e03a6caf09539d + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + license: LGPL-2.0-or-later + license_family: LGPL + size: 99596 + timestamp: 1755102025473 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gsl-2.7-he838d99_0.tar.bz2 + sha256: 132a918b676dd1f533d7c6f95e567abf7081a6ea3251c3280de35ef600e0da87 + md5: fec079ba39c9cca093bf4c00001825de + depends: + - libblas >=3.8.0,<4.0a0 + - libcblas >=3.8.0,<4.0a0 + - libgcc-ng >=9.3.0 + license: GPL-3.0-or-later + license_family: GPL + size: 3376423 + timestamp: 1626369596591 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gst-plugins-base-1.24.11-h651a532_0.conda + sha256: a497d2ba34fdfa4bead423cba5261b7e619df3ac491fb0b6231d91da45bd05fc + md5: d8d8894f8ced2c9be76dc9ad1ae531ce + depends: + - __glibc >=2.17,<3.0.a0 + - alsa-lib >=1.2.14,<1.3.0a0 + - gstreamer 1.24.11 hc37bda9_0 + - libdrm >=2.4.124,<2.5.0a0 + - libegl >=1.7.0,<2.0a0 + - libexpat >=2.7.0,<3.0a0 + - libgcc >=13 + - libgl >=1.7.0,<2.0a0 + - libglib >=2.84.1,<3.0a0 + - libogg >=1.3.5,<1.4.0a0 + - libopus >=1.5.2,<2.0a0 + - libpng >=1.6.47,<1.7.0a0 + - libstdcxx >=13 + - libvorbis >=1.3.7,<1.4.0a0 + - libxcb >=1.17.0,<2.0a0 + - libzlib >=1.3.1,<2.0a0 + - xorg-libx11 >=1.8.12,<2.0a0 + - xorg-libxau >=1.0.12,<2.0a0 + - xorg-libxdamage >=1.1.6,<2.0a0 + - xorg-libxext >=1.3.6,<2.0a0 + - xorg-libxfixes >=6.0.1,<7.0a0 + - xorg-libxrender >=0.9.12,<0.10.0a0 + - xorg-libxshmfence >=1.3.3,<2.0a0 + - xorg-libxxf86vm >=1.1.6,<2.0a0 + license: LGPL-2.0-or-later + license_family: LGPL + size: 2859572 + timestamp: 1745093626455 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gstreamer-1.24.11-hc37bda9_0.conda + sha256: 6e93b99d77ac7f7b3eb29c1911a0a463072a40748b96dbe37c18b2c0a90b34de + md5: 056d86cacf2b48c79c6a562a2486eb8c + depends: + - __glibc >=2.17,<3.0.a0 + - glib >=2.84.1,<3.0a0 + - libgcc >=13 + - libglib >=2.84.1,<3.0a0 + - libiconv >=1.18,<2.0a0 + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + license: LGPL-2.0-or-later + license_family: LGPL + size: 2021832 + timestamp: 1745093493354 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/gxx_impl_linux-64-15.2.0-h54ccb8d_7.conda + sha256: 1fb7da99bcdab2ef8bd2458d8116600524207f3177d5c786d18f3dc5f824a4b8 + md5: f2da2e9e5b7c485f5a4344d5709d8633 + depends: + - gcc_impl_linux-64 15.2.0 hcacfade_7 + - libstdcxx-devel_linux-64 15.2.0 h73f6952_107 + - sysroot_linux-64 + - tzdata + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 16283256 + timestamp: 1759968538523 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/h2-4.3.0-pyhcf101f3_0.conda + sha256: 84c64443368f84b600bfecc529a1194a3b14c3656ee2e832d15a20e0329b6da3 + md5: 164fc43f0b53b6e3a7bc7dce5e4f1dc9 + depends: + - python >=3.10 + - hyperframe >=6.1,<7 + - hpack >=4.1,<5 + - python + license: MIT + license_family: MIT + size: 95967 + timestamp: 1756364871835 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/harfbuzz-12.2.0-h15599e2_0.conda + sha256: 6bd8b22beb7d40562b2889dc68232c589ff0d11a5ad3addd41a8570d11f039d9 + md5: b8690f53007e9b5ee2c2178dd4ac778c + depends: + - __glibc >=2.17,<3.0.a0 + - cairo >=1.18.4,<2.0a0 + - graphite2 >=1.3.14,<2.0a0 + - icu >=75.1,<76.0a0 + - libexpat >=2.7.1,<3.0a0 + - libfreetype >=2.14.1 + - libfreetype6 >=2.14.1 + - libgcc >=14 + - libglib >=2.86.1,<3.0a0 + - libstdcxx >=14 + - libzlib >=1.3.1,<2.0a0 + license: MIT + license_family: MIT + size: 2411408 + timestamp: 1762372726141 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/hdf5-1.14.3-nompi_h2d575fe_109.conda + sha256: e8669a6d76d415f4fdbe682507ac3a3b39e8f493d2f2bdc520817f80b7cc0753 + md5: e7a7a6e6f70553a31e6e79c65768d089 + depends: + - __glibc >=2.17,<3.0.a0 + - libaec >=1.1.3,<2.0a0 + - libcurl >=8.11.1,<9.0a0 + - libgcc >=13 + - libgfortran + - libgfortran5 >=13.3.0 + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + - openssl >=3.4.0,<4.0a0 + license: BSD-3-Clause + license_family: BSD + size: 3930078 + timestamp: 1737516601132 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/hisat2-2.2.1-h503566f_8.tar.bz2 + sha256: c413a90c535a385e7bad2bc8bd968ff2db185b5850b8fbbf214fc97602d57642 + md5: febdedf65aa5a9dc028860df21afc9a6 + depends: + - libgcc >=13 + - libstdcxx >=13 + - perl + - python >3.5 + license: GPL-3.0 + license_family: GPL + size: 16680452 + timestamp: 1734019722585 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/hmmer-3.4-hb6cb901_4.tar.bz2 + sha256: 604f55ee1f16a9f3c2d07f90eeb4331f5b9c1dca8ed1b70e0fea3bf81b140310 + md5: 689f962720e131fe4849e2909181120a + depends: + - gsl >=2.7,<2.8.0a0 + - libgcc >=13 + - openmpi >=4.1.6,<5.0a0 + license: BSD + license_family: BSD + size: 11913715 + timestamp: 1746006254153 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/hpack-4.1.0-pyhd8ed1ab_0.conda + sha256: 6ad78a180576c706aabeb5b4c8ceb97c0cb25f1e112d76495bff23e3779948ba + md5: 0a802cb9888dd14eeefc611f05c40b6e + depends: + - python >=3.9 + license: MIT + license_family: MIT + size: 30731 + timestamp: 1737618390337 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/htslib-1.22.1-h566b1c6_0.tar.bz2 + sha256: 858e634fea447555dcf9725c6533ec7e0f345cb74b53d0d6c99f0ff9575924a1 + md5: 646b491ab86e0ab7173d35eb3a5f3241 + depends: + - bzip2 >=1.0.8,<2.0a0 + - libcurl >=8.14.1,<9.0a0 + - libdeflate >=1.22,<1.26.0a0 + - libgcc >=13 + - liblzma >=5.8.1,<6.0a0 + - libzlib >=1.3.1,<2.0a0 + - openssl >=3.5.1,<4.0a0 + license: MIT + license_family: MIT + size: 3255846 + timestamp: 1752522753233 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/hyperframe-6.1.0-pyhd8ed1ab_0.conda + sha256: 77af6f5fe8b62ca07d09ac60127a30d9069fdc3c68d6b256754d0ffb1f7779f8 + md5: 8e6923fc12f1fe8f8c4e5c9f343256ac + depends: + - python >=3.9 + license: MIT + license_family: MIT + size: 17397 + timestamp: 1737618427549 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/icu-75.1-he02047a_0.conda + sha256: 71e750d509f5fa3421087ba88ef9a7b9be11c53174af3aa4d06aff4c18b38e8e + md5: 8b189310083baabfb622af68fd9d3ae3 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc-ng >=12 + - libstdcxx-ng >=12 + license: MIT + license_family: MIT + size: 12129203 + timestamp: 1720853576813 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/idna-3.11-pyhd8ed1ab_0.conda + sha256: ae89d0299ada2a3162c2614a9d26557a92aa6a77120ce142f8e0109bbf0342b0 + md5: 53abe63df7e10a6ba605dc5f9f961d36 + depends: + - python >=3.10 + license: BSD-3-Clause + license_family: BSD + size: 50721 + timestamp: 1760286526795 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/iqtree-3.0.1-h503566f_0.tar.bz2 + sha256: e0dcb9e8678bd5a50c2c966d1916f9b38fc29ee5e98ce3207e0f50c2145d6dba + md5: fd4dfb10aac70986b4b51d73c45b4312 + depends: + - _openmp_mutex >=4.5 + - libgcc >=13 + - libgomp + - libstdcxx >=13 + license: GPL-2.0-or-later + license_family: GPL2 + size: 4400455 + timestamp: 1752022486560 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/isa-l-2.31.1-hb9d3cd8_1.conda + sha256: 75b15f01a6b286630c4a98be0d05e286275a5ef3868e23e6d9644e51b73650e1 + md5: 00f364ec0a7e975ec9d2fc720b19c129 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: BSD-3-Clause + license_family: BSD + size: 157291 + timestamp: 1736497194571 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/joblib-1.5.2-pyhd8ed1ab_0.conda + sha256: 6fc414c5ae7289739c2ba75ff569b79f72e38991d61eb67426a8a4b92f90462c + md5: 4e717929cfa0d49cef92d911e31d0e90 + depends: + - python >=3.10 + - setuptools + license: BSD-3-Clause + license_family: BSD + size: 224671 + timestamp: 1756321850584 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/jq-1.8.1-h73b1eb8_0.conda + sha256: ab26cb11ad0d10f5c6637d925b044c74a3eacb5825686d3720313b3cb6d40cef + md5: 2714e43bfc035f7ef26796632aa1b523 + depends: + - oniguruma 6.9.* + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + - oniguruma >=6.9.10,<6.10.0a0 + license: MIT + license_family: MIT + size: 313184 + timestamp: 1751447310552 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/jsoncpp-1.9.6-hf42df4d_1.conda + sha256: ed4b1878be103deb2e4c6d0eea3c9bdddfd7fc3178383927dce7578fb1063520 + md5: 7bdc5e2cc11cb0a0f795bdad9732b0f2 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libstdcxx >=13 + license: LicenseRef-Public-Domain OR MIT + size: 169093 + timestamp: 1733780223643 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/jsonpatch-1.33-pyhd8ed1ab_1.conda + sha256: 304955757d1fedbe344af43b12b5467cca072f83cce6109361ba942e186b3993 + md5: cb60ae9cf02b9fcb8004dec4089e5691 + depends: + - jsonpointer >=1.9 + - python >=3.9 + license: BSD-3-Clause + license_family: BSD + size: 17311 + timestamp: 1733814664790 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/jsonpointer-3.0.0-py311h38be061_2.conda + sha256: 4e744b30e3002b519c48868b3f5671328274d1d78cc8cbc0cda43057b570c508 + md5: 5dd29601defbcc14ac6953d9504a80a7 + depends: + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: BSD-3-Clause + license_family: BSD + size: 18368 + timestamp: 1756754243123 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/julia-1.12.1-h212faf0_0.conda + sha256: 4537b6ece3e32e9a316f92f77bd84346bb6ad6ef97511381e58a267a87748ece + md5: dc190fbdcb5f4e6ed4d739e3153a1d69 + depends: + - __glibc >=2.17,<3.0.a0 + - arpack >=3.9.1,<3.10.0a0 nompi_* + - curl + - git + - gmp >=6.3.0,<7.0a0 + - libamd >=3.3.3,<4.0a0 + - libbtf >=2.3.2,<3.0a0 + - libcamd >=3.3.3,<4.0a0 + - libccolamd >=3.3.4,<4.0a0 + - libcholmod >=5.3.1,<6.0a0 + - libcolamd >=3.3.4,<4.0a0 + - libcxsparse >=4.4.1,<5.0a0 + - libgcc >=14 + - libgfortran + - libgfortran5 >=14.3.0 + - libgit2 >=1.9.1,<1.10.0a0 + - libklu >=2.3.5,<3.0a0 + - libldl >=3.3.2,<4.0a0 + - libnghttp2 >=1.67.0,<2.0a0 + - libopenlibm4 >=0.8.1,<1.0a0 + - libparu >=1.0.0,<2.0a0 + - librbio >=4.3.4,<5.0a0 + - libspex >=3.2.3,<4.0a0 + - libspqr >=4.3.4,<5.0a0 + - libssh2 >=1.11.1,<2.0a0 + - libstdcxx >=14 + - libsuitesparseconfig >=7.10.1,<8.0a0 + - libumfpack >=6.3.5,<7.0a0 + - libunwind >=1.6.2,<1.7.0a0 + - libutf8proc >=2.11.0,<2.12.0a0 + - libzlib >=1.3.1,<2.0a0 + - mpfr >=4.2.1,<5.0a0 + - openblas-ilp64 + - openlibm + - p7zip + - pcre2 >=10.46,<10.47.0a0 + - suitesparse >=7.10.1,<8.0a0 + - zlib + license: MIT + license_family: MIT + size: 171812998 + timestamp: 1762176763294 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/kallisto-0.51.1-h2b92561_2.tar.bz2 + sha256: dec322eb26df26ce8dac69fb058bb5328b34876d328f524dc900f127a7eabe2a + md5: b76c20fc5d0a865b21eb8b384cadbae8 + depends: + - bzip2 >=1.0.8,<2.0a0 + - hdf5 >=1.14.3,<1.14.4.0a0 + - libcurl >=8.14.1,<9.0a0 + - libgcc >=13 + - liblzma >=5.8.1,<6.0a0 + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + - zlib-ng >=2.2.5,<2.3.0a0 + license: BSD-2-Clause + license_family: BSD + size: 1085690 + timestamp: 1754786122034 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/kernel-headers_linux-64-4.18.0-he073ed8_8.conda + sha256: 305c22a251db227679343fd73bfde121e555d466af86e537847f4c8b9436be0d + md5: ff007ab0f0fdc53d245972bba8a6d40c + constrains: + - sysroot_linux-64 ==2.28 + license: LGPL-2.0-or-later AND LGPL-2.0-or-later WITH exceptions AND GPL-2.0-or-later + license_family: GPL + size: 1272697 + timestamp: 1752669126073 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/keyutils-1.6.3-hb9d3cd8_0.conda + sha256: 0960d06048a7185d3542d850986d807c6e37ca2e644342dd0c72feefcf26c2a4 + md5: b38117a3c920364aff79f870c984b4a3 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: LGPL-2.1-or-later + size: 134088 + timestamp: 1754905959823 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/kiwisolver-1.4.9-py311h724c32c_2.conda + sha256: 81181e88c0d49cc86bc687e2583da0cb0b651525bf17d4f4f3aecb1596441769 + md5: 4089f739463c798e10d8644bc34e24de + depends: + - python + - libstdcxx >=14 + - libgcc >=14 + - __glibc >=2.17,<3.0.a0 + - python_abi 3.11.* *_cp311 + license: BSD-3-Clause + size: 78452 + timestamp: 1762488745068 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/kmer-jellyfish-2.3.1-py311pl5321he264feb_6.tar.bz2 + sha256: d2fd8d9201dd4fa21b14f2394024869e4ed38c6e4a13041cd5df5423367ac755 + md5: 710d637c78780cf668a19e859403b16a + depends: + - libgcc >=13 + - libstdcxx >=13 + - perl >=5.32.1,<5.33.0a0 *_perl5 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: GPL-3.0-or-later + license_family: GPL3 + size: 643575 + timestamp: 1748328589655 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/krb5-1.21.3-h659f571_0.conda + sha256: 99df692f7a8a5c27cd14b5fb1374ee55e756631b9c3d659ed3ee60830249b238 + md5: 3f43953b7d3fb3aaa1d0d0723d91e368 + depends: + - keyutils >=1.6.1,<2.0a0 + - libedit >=3.1.20191231,<3.2.0a0 + - libedit >=3.1.20191231,<4.0a0 + - libgcc-ng >=12 + - libstdcxx-ng >=12 + - openssl >=3.3.1,<4.0a0 + license: MIT + license_family: MIT + size: 1370023 + timestamp: 1719463201255 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lame-3.100-h166bdaf_1003.tar.bz2 + sha256: aad2a703b9d7b038c0f745b853c6bb5f122988fe1a7a096e0e606d9cbec4eaab + md5: a8832b479f93521a9e7b5b743803be51 + depends: + - libgcc-ng >=12 + license: LGPL-2.0-only + license_family: LGPL + size: 508258 + timestamp: 1664996250081 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lcms2-2.17-h717163a_0.conda + sha256: d6a61830a354da022eae93fa896d0991385a875c6bba53c82263a289deda9db8 + md5: 000e85703f0fd9594c81710dd5066471 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libjpeg-turbo >=3.0.0,<4.0a0 + - libtiff >=4.7.0,<4.8.0a0 + license: MIT + license_family: MIT + size: 248046 + timestamp: 1739160907615 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/ld_impl_linux-64-2.45-h1aa0949_0.conda + sha256: 32321d38b8785ef8ddcfef652ee370acee8d944681014d47797a18637ff16854 + md5: 1450224b3e7d17dfeb985364b77a4d47 + depends: + - __glibc >=2.17,<3.0.a0 + - zstd >=1.5.7,<1.6.0a0 + constrains: + - binutils_impl_linux-64 2.45 + license: GPL-3.0-only + size: 753744 + timestamp: 1763060439129 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lerc-4.0.0-h0aef613_1.conda + sha256: 412381a43d5ff9bbed82cd52a0bbca5b90623f62e41007c9c42d3870c60945ff + md5: 9344155d33912347b37f0ae6c410a835 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libstdcxx >=13 + license: Apache-2.0 + license_family: Apache + size: 264243 + timestamp: 1745264221534 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libaec-1.1.4-h3f801dc_0.conda + sha256: 410ab78fe89bc869d435de04c9ffa189598ac15bb0fe1ea8ace8fb1b860a2aa3 + md5: 01ba04e414e47f95c03d6ddd81fd37be + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libstdcxx >=13 + license: BSD-2-Clause + license_family: BSD + size: 36825 + timestamp: 1749993532943 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libamd-3.3.3-h456b2da_7100101.conda + sha256: 5fc32a5497c9919ffde729a604b0acfa97c403ce5b2b27b28ca261cf0c4643aa + md5: a067596d679bcde85375143e7c374738 + depends: + - __glibc >=2.17,<3.0.a0 + - libgfortran5 >=13.3.0 + - libgfortran + - libgcc >=13 + - libsuitesparseconfig >=7.10.1,<8.0a0 + license: BSD-3-Clause + license_family: BSD + size: 48250 + timestamp: 1741963824815 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libarchive-3.8.1-gpl_h98cc613_100.conda + sha256: 6f35e429909b0fa6a938f8ff79e1d7000e8f15fbb37f67be6f789348fea4c602 + md5: 9de6247361e1ee216b09cfb8b856e2ee + depends: + - __glibc >=2.17,<3.0.a0 + - bzip2 >=1.0.8,<2.0a0 + - libgcc >=13 + - liblzma >=5.8.1,<6.0a0 + - libxml2 >=2.13.8,<2.14.0a0 + - libzlib >=1.3.1,<2.0a0 + - lz4-c >=1.10.0,<1.11.0a0 + - lzo >=2.10,<3.0a0 + - openssl >=3.5.0,<4.0a0 + - zstd >=1.5.7,<1.6.0a0 + license: BSD-2-Clause + license_family: BSD + size: 883383 + timestamp: 1749385818314 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libasprintf-0.25.1-h3f43e3d_1.conda + sha256: cb728a2a95557bb6a5184be2b8be83a6f2083000d0c7eff4ad5bbe5792133541 + md5: 3b0d184bc9404516d418d4509e418bdc + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + license: LGPL-2.1-or-later + size: 53582 + timestamp: 1753342901341 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libasprintf-devel-0.25.1-h3f43e3d_1.conda + sha256: 2fc95060efc3d76547b7872875af0b7212d4b1407165be11c5f830aeeb57fc3a + md5: fd9cf4a11d07f0ef3e44fc061611b1ed + depends: + - __glibc >=2.17,<3.0.a0 + - libasprintf 0.25.1 h3f43e3d_1 + - libgcc >=14 + license: LGPL-2.1-or-later + size: 34734 + timestamp: 1753342921605 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libblas-3.9.0-38_h4a7cf45_openblas.conda + build_number: 38 + sha256: b26a32302194e05fa395d5135699fd04a905c6ad71f24333f97c64874e053623 + md5: 3509b5e2aaa5f119013c8969fdd9a905 + depends: + - libopenblas >=0.3.30,<0.3.31.0a0 + - libopenblas >=0.3.30,<1.0a0 + constrains: + - libcblas 3.9.0 38*_openblas + - blas 2.138 openblas + - liblapacke 3.9.0 38*_openblas + - mkl <2026 + - liblapack 3.9.0 38*_openblas + license: BSD-3-Clause + license_family: BSD + size: 17522 + timestamp: 1761680084434 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libboost-1.85.0-h0ccab89_4.conda + sha256: dc19dfc636c363871763384219269ce6a027fcf3831f17e018caeecb2ffbb20a + md5: 4da1690badd566fc1041f91cd5655727 + depends: + - __glibc >=2.17,<3.0.a0 + - bzip2 >=1.0.8,<2.0a0 + - icu >=75.1,<76.0a0 + - libgcc-ng >=12 + - libstdcxx-ng >=12 + - libzlib >=1.3.1,<2.0a0 + - xz >=5.2.6,<6.0a0 + - zstd >=1.5.6,<1.6.0a0 + constrains: + - boost-cpp =1.85.0 + license: BSL-1.0 + size: 2869710 + timestamp: 1722289756758 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libboost-devel-1.85.0-h00ab1b0_4.conda + sha256: 04ec5a59e87d75cf6f8b539493f7f71c0cca3f50976251895f51da45e39cddf7 + md5: ded76b8670cb505006c891c4d45844a5 + depends: + - libboost 1.85.0 h0ccab89_4 + - libboost-headers 1.85.0 ha770c72_4 + constrains: + - boost-cpp =1.85.0 + license: BSL-1.0 + size: 40881 + timestamp: 1722289871820 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libboost-headers-1.85.0-ha770c72_4.conda + sha256: 55aa2ac604bd7ed76fae0c93698d37aae455ca2fb229ff9aa45e085ff7ad48ec + md5: 00e4848983222729ccb7c69f1039f4b9 + constrains: + - boost-cpp =1.85.0 + license: BSL-1.0 + size: 13961521 + timestamp: 1722289776587 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libbrotlicommon-1.2.0-h09219d5_0.conda + sha256: fbbcd11742bb8c96daa5f4f550f1804a902708aad2092b39bec3faaa2c8ae88a + md5: 9b3117ec960b823815b02190b41c0484 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + license: MIT + license_family: MIT + size: 79664 + timestamp: 1761592192478 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libbrotlidec-1.2.0-hd53d788_0.conda + sha256: f7f357c33bd10afd58072ad4402853a8522d52d00d7ae9adb161ecf719f63574 + md5: c183787d2b228775dece45842abbbe53 + depends: + - __glibc >=2.17,<3.0.a0 + - libbrotlicommon 1.2.0 h09219d5_0 + - libgcc >=14 + license: MIT + license_family: MIT + size: 34445 + timestamp: 1761592202559 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libbrotlienc-1.2.0-h02bd7ab_0.conda + sha256: 1370c8b1a215751c4592bf95d4b5d11bac91c577770efcb237e3a0f35c326559 + md5: b7a924e3e9ebc7938ffc7d94fe603ed3 + depends: + - __glibc >=2.17,<3.0.a0 + - libbrotlicommon 1.2.0 h09219d5_0 + - libgcc >=14 + license: MIT + license_family: MIT + size: 298252 + timestamp: 1761592214576 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libbtf-2.3.2-hf02c80a_7100101.conda + sha256: fe36f414f48ab87251f02aeef1fcbb6f3929322316842dada0f8142db2710264 + md5: 6f4aec52002defbdf3e24eb79e56a209 + depends: + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + - libsuitesparseconfig >=7.10.1,<8.0a0 + license: LGPL-2.1-or-later + size: 26913 + timestamp: 1741963824815 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcamd-3.3.3-hf02c80a_7100101.conda + sha256: 16e9ae4e173a8606b0b8be118dbdcf4e03c9dd9777eea6bf9dff4397133d0d06 + md5: 1c9d1532caadece8adc2d14c6d4fc726 + depends: + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + - libsuitesparseconfig >=7.10.1,<8.0a0 + license: BSD-3-Clause + license_family: BSD + size: 44119 + timestamp: 1741963824815 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcap-2.77-h3ff7636_0.conda + sha256: 9517cce5193144af0fcbf19b7bd67db0a329c2cc2618f28ffecaa921a1cbe9d3 + md5: 09c264d40c67b82b49a3f3b89037bd2e + depends: + - __glibc >=2.17,<3.0.a0 + - attr >=2.5.2,<2.6.0a0 + - libgcc >=14 + license: BSD-3-Clause + license_family: BSD + size: 121429 + timestamp: 1762349484074 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcblas-3.9.0-38_h0358290_openblas.conda + build_number: 38 + sha256: 7fe653f45c01eb16d7b48ad934b068dad2885d6f4a7c41512b6a5f1f522bffe9 + md5: bcd928a9376a215cd9164a4312dd5e98 + depends: + - libblas 3.9.0 38_h4a7cf45_openblas + constrains: + - blas 2.138 openblas + - liblapack 3.9.0 38*_openblas + - liblapacke 3.9.0 38*_openblas + license: BSD-3-Clause + license_family: BSD + size: 17503 + timestamp: 1761680091587 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libccolamd-3.3.4-hf02c80a_7100101.conda + sha256: cc90aa5e0ad1f7ae9a29d9a42aacd7f7f02aba0bf5467513bfda7e6b18a4cbc8 + md5: e5107e02dc4c2f9f41eef72d72c23517 + depends: + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + - libsuitesparseconfig >=7.10.1,<8.0a0 + license: BSD-3-Clause + license_family: BSD + size: 41578 + timestamp: 1741963824815 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcholmod-5.3.1-h9cf07ce_7100101.conda + sha256: 69540315b4b8de93b383243334151ed19e98968baaa59440ba645a3bff68d765 + md5: f51e24ce110ae24c92074736a308e47e + depends: + - libgcc >=13 + - libstdcxx >=13 + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + - _openmp_mutex >=4.5 + - liblapack >=3.9.0,<4.0a0 + - libcolamd >=3.3.4,<4.0a0 + - libamd >=3.3.3,<4.0a0 + - libsuitesparseconfig >=7.10.1,<8.0a0 + - libccolamd >=3.3.4,<4.0a0 + - libblas >=3.9.0,<4.0a0 + - libcamd >=3.3.3,<4.0a0 + license: LGPL-2.1-or-later AND GPL-2.0-or-later AND Apache-2.0 + size: 990886 + timestamp: 1741963824815 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libclang-cpp20.1-20.1.8-default_h99862b1_4.conda + sha256: be2cd2768c932ade04bc4868b0f564ce6681bc861f28027419dc6651525afeb1 + md5: 2a7f3bca5b60a34be5a35cbc70711bce + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libllvm20 >=20.1.8,<20.2.0a0 + - libstdcxx >=14 + license: Apache-2.0 WITH LLVM-exception + license_family: Apache + size: 21250739 + timestamp: 1759440009094 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libclang-cpp21.1-21.1.0-default_h99862b1_1.conda + sha256: efe9f1363a49668d10aacdb8be650433fab659f05ed6cc2b9da00e3eb7eaf602 + md5: d599b346638b9216c1e8f9146713df05 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libllvm21 >=21.1.0,<21.2.0a0 + - libstdcxx >=14 + license: Apache-2.0 WITH LLVM-exception + license_family: Apache + size: 21131028 + timestamp: 1757383135034 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libclang13-21.1.0-default_h746c552_1.conda + sha256: e6c0123b888d6abf03c66c52ed89f9de1798dde930c5fd558774f26e994afbc6 + md5: 327c78a8ce710782425a89df851392f7 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libllvm21 >=21.1.0,<21.2.0a0 + - libstdcxx >=14 + license: Apache-2.0 WITH LLVM-exception + license_family: Apache + size: 12358102 + timestamp: 1757383373129 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcolamd-3.3.4-hf02c80a_7100101.conda + sha256: 00d1b976b914f0c20ae6f81f4e4713fa87717542eba8757b9a3c9e8abcc29858 + md5: 56d4c5542887e8955f21f8546ad75d9d + depends: + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + - libsuitesparseconfig >=7.10.1,<8.0a0 + license: BSD-3-Clause + license_family: BSD + size: 33160 + timestamp: 1741963824815 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcups-2.3.3-hb8b1518_5.conda + sha256: cb83980c57e311783ee831832eb2c20ecb41e7dee6e86e8b70b8cef0e43eab55 + md5: d4a250da4737ee127fb1fa6452a9002e + depends: + - __glibc >=2.17,<3.0.a0 + - krb5 >=1.21.3,<1.22.0a0 + - libgcc >=13 + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + license: Apache-2.0 + license_family: Apache + size: 4523621 + timestamp: 1749905341688 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcurl-8.17.0-h4e3cde8_0.conda + sha256: 100e29ca864c32af15a5cc354f502d07b2600218740fdf2439fa7d66b50b3529 + md5: 01e149d4a53185622dc2e788281961f2 + depends: + - __glibc >=2.17,<3.0.a0 + - krb5 >=1.21.3,<1.22.0a0 + - libgcc >=14 + - libnghttp2 >=1.67.0,<2.0a0 + - libssh2 >=1.11.1,<2.0a0 + - libzlib >=1.3.1,<2.0a0 + - openssl >=3.5.4,<4.0a0 + - zstd >=1.5.7,<1.6.0a0 + license: curl + license_family: MIT + size: 460366 + timestamp: 1762333743748 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libcxsparse-4.4.1-hf02c80a_7100101.conda + sha256: ab40fc8a4662f550d053576a56db896247bc81eb291eff3811f24c231829e3dd + md5: 917931d508582ef891bbac172294d9fb + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - _openmp_mutex >=4.5 + - libsuitesparseconfig >=7.10.1,<8.0a0 + license: LGPL-2.1-or-later + size: 113979 + timestamp: 1741963824815 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libdb-6.2.32-h9c3ff4c_0.tar.bz2 + sha256: 21fac1012ff05b131d4b5d284003dbbe7b5c4c652aa9e401b46279ed5a784372 + md5: 3f3258d8f841fbac63b36b75bdac1afd + depends: + - libgcc-ng >=9.3.0 + - libstdcxx-ng >=9.3.0 + license: AGPL-3.0-only + license_family: AGPL + size: 24409456 + timestamp: 1609539093147 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libdeflate-1.25-h17f619e_0.conda + sha256: aa8e8c4be9a2e81610ddf574e05b64ee131fab5e0e3693210c9d6d2fba32c680 + md5: 6c77a605a7a689d17d4819c0f8ac9a00 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + license: MIT + license_family: MIT + size: 73490 + timestamp: 1761979956660 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libdrm-2.4.125-hb03c661_1.conda + sha256: c076a213bd3676cc1ef22eeff91588826273513ccc6040d9bea68bccdc849501 + md5: 9314bc5a1fe7d1044dc9dfd3ef400535 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libpciaccess >=0.18,<0.19.0a0 + license: MIT + license_family: MIT + size: 310785 + timestamp: 1757212153962 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libedit-3.1.20250104-pl5321h7949ede_0.conda + sha256: d789471216e7aba3c184cd054ed61ce3f6dac6f87a50ec69291b9297f8c18724 + md5: c277e0a4d549b03ac1e9d6cbbe3d017b + depends: + - ncurses + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - ncurses >=6.5,<7.0a0 + license: BSD-2-Clause + license_family: BSD + size: 134676 + timestamp: 1738479519902 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libegl-1.7.0-ha4b6fd6_2.conda + sha256: 7fd5408d359d05a969133e47af580183fbf38e2235b562193d427bb9dad79723 + md5: c151d5eb730e9b7480e6d48c0fc44048 + depends: + - __glibc >=2.17,<3.0.a0 + - libglvnd 1.7.0 ha4b6fd6_2 + license: LicenseRef-libglvnd + size: 44840 + timestamp: 1731330973553 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libev-4.33-hd590300_2.conda + sha256: 1cd6048169fa0395af74ed5d8f1716e22c19a81a8a36f934c110ca3ad4dd27b4 + md5: 172bf1cd1ff8629f2b1179945ed45055 + depends: + - libgcc-ng >=12 + license: BSD-2-Clause + license_family: BSD + size: 112766 + timestamp: 1702146165126 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libevent-2.1.12-hf998b51_1.conda + sha256: 2e14399d81fb348e9d231a82ca4d816bf855206923759b69ad006ba482764131 + md5: a1cfcc585f0c42bf8d5546bb1dfb668d + depends: + - libgcc-ng >=12 + - openssl >=3.1.1,<4.0a0 + license: BSD-3-Clause + license_family: BSD + size: 427426 + timestamp: 1685725977222 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libexpat-2.7.1-hecca717_0.conda + sha256: da2080da8f0288b95dd86765c801c6e166c4619b910b11f9a8446fb852438dc2 + md5: 4211416ecba1866fab0c6470986c22d6 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + constrains: + - expat 2.7.1.* + license: MIT + license_family: MIT + size: 74811 + timestamp: 1752719572741 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libffi-3.5.2-h9ec8514_0.conda + sha256: 25cbdfa65580cfab1b8d15ee90b4c9f1e0d72128f1661449c9a999d341377d54 + md5: 35f29eec58405aaf55e01cb470d8c26a + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + license: MIT + license_family: MIT + size: 57821 + timestamp: 1760295480630 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libflac-1.4.3-h59595ed_0.conda + sha256: 65908b75fa7003167b8a8f0001e11e58ed5b1ef5e98b96ab2ba66d7c1b822c7d + md5: ee48bf17cc83a00f59ca1494d5646869 + depends: + - gettext >=0.21.1,<1.0a0 + - libgcc-ng >=12 + - libogg 1.3.* + - libogg >=1.3.4,<1.4.0a0 + - libstdcxx-ng >=12 + license: BSD-3-Clause + license_family: BSD + size: 394383 + timestamp: 1687765514062 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libfreetype-2.14.1-ha770c72_0.conda + sha256: 4641d37faeb97cf8a121efafd6afd040904d4bca8c46798122f417c31d5dfbec + md5: f4084e4e6577797150f9b04a4560ceb0 + depends: + - libfreetype6 >=2.14.1 + license: GPL-2.0-only OR FTL + size: 7664 + timestamp: 1757945417134 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libfreetype6-2.14.1-h73754d4_0.conda + sha256: 4a7af818a3179fafb6c91111752954e29d3a2a950259c14a2fc7ba40a8b03652 + md5: 8e7251989bca326a28f4a5ffbd74557a + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libpng >=1.6.50,<1.7.0a0 + - libzlib >=1.3.1,<2.0a0 + constrains: + - freetype >=2.14.1 + license: GPL-2.0-only OR FTL + size: 386739 + timestamp: 1757945416744 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgcc-15.2.0-h767d61c_7.conda + sha256: 08f9b87578ab981c7713e4e6a7d935e40766e10691732bba376d4964562bcb45 + md5: c0374badb3a5d4b1372db28d19462c53 + depends: + - __glibc >=2.17,<3.0.a0 + - _openmp_mutex >=4.5 + constrains: + - libgomp 15.2.0 h767d61c_7 + - libgcc-ng ==15.2.0=*_7 + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 822552 + timestamp: 1759968052178 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/libgcc-devel_linux-64-15.2.0-h73f6952_107.conda + sha256: 67323768cddb87e744d0e593f92445cd10005e04259acd3e948c7ba3bcb03aed + md5: 85fce551e54a1e81b69f9ffb3ade6aee + depends: + - __unix + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 2728965 + timestamp: 1759967882886 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgcc-ng-15.2.0-h69a702a_7.conda + sha256: 2045066dd8e6e58aaf5ae2b722fb6dfdbb57c862b5f34ac7bfb58c40ef39b6ad + md5: 280ea6eee9e2ddefde25ff799c4f0363 + depends: + - libgcc 15.2.0 h767d61c_7 + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 29313 + timestamp: 1759968065504 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgettextpo-0.25.1-h3f43e3d_1.conda + sha256: 50a9e9815cf3f5bce1b8c5161c0899cc5b6c6052d6d73a4c27f749119e607100 + md5: 2f4de899028319b27eb7a4023be5dfd2 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libiconv >=1.18,<2.0a0 + license: GPL-3.0-or-later + license_family: GPL + size: 188293 + timestamp: 1753342911214 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgettextpo-devel-0.25.1-h3f43e3d_1.conda + sha256: c7ea10326fd450a2a21955987db09dde78c99956a91f6f05386756a7bfe7cc04 + md5: 3f7a43b3160ec0345c9535a9f0d7908e + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libgettextpo 0.25.1 h3f43e3d_1 + - libiconv >=1.18,<2.0a0 + license: GPL-3.0-or-later + license_family: GPL + size: 37407 + timestamp: 1753342931100 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgfortran-15.2.0-h69a702a_7.conda + sha256: 9ca24328e31c8ef44a77f53104773b9fe50ea8533f4c74baa8489a12de916f02 + md5: 8621a450add4e231f676646880703f49 + depends: + - libgfortran5 15.2.0 hcd61629_7 + constrains: + - libgfortran-ng ==15.2.0=*_7 + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 29275 + timestamp: 1759968110483 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgfortran-ng-15.2.0-h69a702a_7.conda + sha256: 63e1d1ce309e3f42a11637ecf0f29b5cf1550ca6d46412edf82e8e249b1917a1 + md5: beeb74a6fe5ff118451cf0581bfe2642 + depends: + - libgfortran 15.2.0 h69a702a_7 + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 29330 + timestamp: 1759968394141 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgfortran5-15.2.0-hcd61629_7.conda + sha256: e93ceda56498d98c9f94fedec3e2d00f717cbedfc97c49be0e5a5828802f2d34 + md5: f116940d825ffc9104400f0d7f1a4551 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=15.2.0 + constrains: + - libgfortran 15.2.0 + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 1572758 + timestamp: 1759968082504 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgit2-1.9.1-h20a291d_1.conda + sha256: f4b8a56366b046a1fa8dfcb3ee231da1dc738930221e37c41db0055614b85274 + md5: 77ade4eb620737ecbb5adf0bb531775d + depends: + - libgcc >=14 + - libstdcxx >=14 + - libgcc >=14 + - __glibc >=2.17,<3.0.a0 + - libssh2 >=1.11.1,<2.0a0 + - openssl >=3.5.3,<4.0a0 + - pcre2 >=10.46,<10.47.0a0 + - libzlib >=1.3.1,<2.0a0 + license: GPL-2.0-only WITH GCC-exception-2.0 + license_family: GPL + size: 1035135 + timestamp: 1758613895831 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgl-1.7.0-ha4b6fd6_2.conda + sha256: dc2752241fa3d9e40ce552c1942d0a4b5eeb93740c9723873f6fcf8d39ef8d2d + md5: 928b8be80851f5d8ffb016f9c81dae7a + depends: + - __glibc >=2.17,<3.0.a0 + - libglvnd 1.7.0 ha4b6fd6_2 + - libglx 1.7.0 ha4b6fd6_2 + license: LicenseRef-libglvnd + size: 134712 + timestamp: 1731330998354 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libglib-2.86.1-h32235b2_2.conda + sha256: fc82277d0d6340743732c48dcbac3f4e9ee36902649a7d9a02622b0713ce3666 + md5: 986dcf488a1aced411da84753d93d078 + depends: + - __glibc >=2.17,<3.0.a0 + - libffi >=3.5.2,<3.6.0a0 + - libgcc >=14 + - libiconv >=1.18,<2.0a0 + - libzlib >=1.3.1,<2.0a0 + - pcre2 >=10.46,<10.47.0a0 + constrains: + - glib 2.86.1 *_2 + license: LGPL-2.1-or-later + size: 3933707 + timestamp: 1762787455198 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libglvnd-1.7.0-ha4b6fd6_2.conda + sha256: 1175f8a7a0c68b7f81962699751bb6574e6f07db4c9f72825f978e3016f46850 + md5: 434ca7e50e40f4918ab701e3facd59a0 + depends: + - __glibc >=2.17,<3.0.a0 + license: LicenseRef-libglvnd + size: 132463 + timestamp: 1731330968309 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libglx-1.7.0-ha4b6fd6_2.conda + sha256: 2d35a679624a93ce5b3e9dd301fff92343db609b79f0363e6d0ceb3a6478bfa7 + md5: c8013e438185f33b13814c5c488acd5c + depends: + - __glibc >=2.17,<3.0.a0 + - libglvnd 1.7.0 ha4b6fd6_2 + - xorg-libx11 >=1.8.10,<2.0a0 + license: LicenseRef-libglvnd + size: 75504 + timestamp: 1731330988898 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libgomp-15.2.0-h767d61c_7.conda + sha256: e9fb1c258c8e66ee278397b5822692527c5f5786d372fe7a869b900853f3f5ca + md5: f7b4d76975aac7e5d9e6ad13845f92fe + depends: + - __glibc >=2.17,<3.0.a0 + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 447919 + timestamp: 1759967942498 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libhwloc-2.12.1-default_h3d81e11_1000.conda + sha256: eecaf76fdfc085d8fed4583b533c10cb7f4a6304be56031c43a107e01a56b7e2 + md5: d821210ab60be56dd27b5525ed18366d + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - libxml2 >=2.13.8,<2.14.0a0 + license: BSD-3-Clause + license_family: BSD + size: 2450422 + timestamp: 1752761850672 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libiconv-1.18-h3b78370_2.conda + sha256: c467851a7312765447155e071752d7bf9bf44d610a5687e32706f480aad2833f + md5: 915f5995e94f60e9a4826e0b0920ee88 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + license: LGPL-2.1-only + size: 790176 + timestamp: 1754908768807 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libidn2-2.3.8-hfac485b_1.conda + sha256: cc38c900b9a20fe75e61cbb594e749c57a06d96510722f5ddfa309682062b065 + md5: 842a81de672ddcf476337c8bde3cad33 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libunistring >=0,<1.0a0 + license: LGPL-2.0-only + license_family: LGPL + size: 139036 + timestamp: 1760385590993 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libjemalloc-5.3.0-h5888daf_1.conda + sha256: 4b43f86519a9a31aaa2473fe8373cf313713ce83e349991b4faa0c0995e8b0bb + md5: 29f738fc41e063bfb120b1a667f684dc + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libstdcxx >=13 + license: BSD-2-Clause + license_family: BSD + size: 1571937 + timestamp: 1726817494251 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libjpeg-turbo-3.1.2-hb03c661_0.conda + sha256: cc9aba923eea0af8e30e0f94f2ad7156e2984d80d1e8e7fe6be5a1f257f0eb32 + md5: 8397539e3a0bbd1695584fb4f927485a + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + constrains: + - jpeg <0.0.0a + license: IJG AND BSD-3-Clause AND Zlib + size: 633710 + timestamp: 1762094827865 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libklu-2.3.5-h95ff59c_7100101.conda + sha256: 6b4d462642c240dc3671af74f7705b23f34eea0f71e0d9dbcf14b4ed008311ff + md5: efaa5e7dc6989363585fbb591480b256 + depends: + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + - _openmp_mutex >=4.5 + - metis >=5.1.0,<5.1.1.0a0 + - libcamd >=3.3.3,<4.0a0 + - liblapack >=3.9.0,<4.0a0 + - libcblas >=3.9.0,<4.0a0 + - libsuitesparseconfig >=7.10.1,<8.0a0 + - libcolamd >=3.3.4,<4.0a0 + - libamd >=3.3.3,<4.0a0 + - libcholmod >=5.3.1,<6.0a0 + - libblas >=3.9.0,<4.0a0 + - libbtf >=2.3.2,<3.0a0 + - libccolamd >=3.3.4,<4.0a0 + license: LGPL-2.1-or-later + size: 131775 + timestamp: 1741963824816 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/liblapack-3.9.0-38_h47877c9_openblas.conda + build_number: 38 + sha256: 63d6073dd4f82ab46943ad99a22fc4edda83b0f8fe6170bdaba7a43352bed007 + md5: 88f10bff57b423a3fd2d990c6055771e + depends: + - libblas 3.9.0 38_h4a7cf45_openblas + constrains: + - libcblas 3.9.0 38*_openblas + - blas 2.138 openblas + - liblapacke 3.9.0 38*_openblas + license: BSD-3-Clause + license_family: BSD + size: 17501 + timestamp: 1761680098660 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libldl-3.3.2-hf02c80a_7100101.conda + sha256: 590232cd302047023ab31b80458833a71b10aeabee7474304dc65db322b5cd70 + md5: 19b71122fea7f6b1c4815f385b2da419 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libsuitesparseconfig >=7.10.1,<8.0a0 + license: LGPL-2.1-or-later + size: 23391 + timestamp: 1741963824815 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libllvm20-20.1.8-hecd9e04_0.conda + sha256: a6fddc510de09075f2b77735c64c7b9334cf5a26900da351779b275d9f9e55e1 + md5: 59a7b967b6ef5d63029b1712f8dcf661 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - libxml2 >=2.13.8,<2.14.0a0 + - libzlib >=1.3.1,<2.0a0 + - zstd >=1.5.7,<1.6.0a0 + license: Apache-2.0 WITH LLVM-exception + license_family: Apache + size: 43987020 + timestamp: 1752141980723 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libllvm21-21.1.0-hecd9e04_0.conda + sha256: d190f1bf322149321890908a534441ca2213a9a96c59819da6cabf2c5b474115 + md5: 9ad637a7ac380c442be142dfb0b1b955 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - libxml2 >=2.13.8,<2.14.0a0 + - libzlib >=1.3.1,<2.0a0 + - zstd >=1.5.7,<1.6.0a0 + license: Apache-2.0 WITH LLVM-exception + license_family: Apache + size: 44363060 + timestamp: 1756291822911 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libltdl-2.4.3a-h5888daf_0.conda + sha256: 7620c6425d4491e17083106ca49624448fc16186c30a93cf2b58f862bba416d1 + md5: 8e5de39cab514fa908fcaa7ba37a8738 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: LGPL-2.0-or-later + license_family: LGPL + size: 38472 + timestamp: 1740593829307 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/liblzma-5.8.1-hb9d3cd8_2.conda + sha256: f2591c0069447bbe28d4d696b7fcb0c5bd0b4ac582769b89addbcf26fb3430d8 + md5: 1a580f7796c7bf6393fddb8bbbde58dc + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + constrains: + - xz 5.8.1.* + license: 0BSD + size: 112894 + timestamp: 1749230047870 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/liblzma-devel-5.8.1-hb9d3cd8_2.conda + sha256: 329e66330a8f9cbb6a8d5995005478188eb4ba8a6b6391affa849744f4968492 + md5: f61edadbb301530bd65a32646bd81552 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - liblzma 5.8.1 hb9d3cd8_2 + license: 0BSD + size: 439868 + timestamp: 1749230061968 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libmamba-2.3.2-hae34dd5_2.conda + sha256: 8d6f1cae31cf9f672e9ccac00d0660bcc13c4457466996247e97e53b1137e68a + md5: b0cfdfd4632cc769a1ca211ecfc1093c + depends: + - cpp-expected >=1.3.1,<1.3.2.0a0 + - libstdcxx >=14 + - libgcc >=14 + - __glibc >=2.17,<3.0.a0 + - yaml-cpp >=0.8.0,<0.9.0a0 + - zstd >=1.5.7,<1.6.0a0 + - nlohmann_json-abi ==3.12.0 + - openssl >=3.5.4,<4.0a0 + - reproc >=14.2,<15.0a0 + - fmt >=11.2.0,<11.3.0a0 + - simdjson >=4.0.7,<4.1.0a0 + - reproc-cpp >=14.2,<15.0a0 + - libarchive >=3.8.1,<3.9.0a0 + - libsolv >=0.7.35,<0.8.0a0 + - libcurl >=8.14.1,<9.0a0 + license: BSD-3-Clause + license_family: BSD + size: 2487190 + timestamp: 1760104375106 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libmambapy-2.3.2-py311h52fc1f4_2.conda + sha256: 0c71fe26a7fa5cdb3e9b284b35dc737f8fd51b1ef4f8b88bc0be1d23a553df21 + md5: bbd22564bc6a3c1ab444cb44da546efc + depends: + - python + - libmamba ==2.3.2 hae34dd5_2 + - __glibc >=2.17,<3.0.a0 + - libstdcxx >=14 + - libgcc >=14 + - nlohmann_json-abi ==3.12.0 + - yaml-cpp >=0.8.0,<0.9.0a0 + - python_abi 3.11.* *_cp311 + - zstd >=1.5.7,<1.6.0a0 + - openssl >=3.5.4,<4.0a0 + - libmamba >=2.3.2,<2.4.0a0 + - fmt >=11.2.0,<11.3.0a0 + - pybind11-abi ==4 + license: BSD-3-Clause + license_family: BSD + size: 768368 + timestamp: 1760104375111 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libnghttp2-1.67.0-had1ee68_0.conda + sha256: a4a7dab8db4dc81c736e9a9b42bdfd97b087816e029e221380511960ac46c690 + md5: b499ce4b026493a13774bcf0f4c33849 + depends: + - __glibc >=2.17,<3.0.a0 + - c-ares >=1.34.5,<2.0a0 + - libev >=4.33,<4.34.0a0 + - libev >=4.33,<5.0a0 + - libgcc >=14 + - libstdcxx >=14 + - libzlib >=1.3.1,<2.0a0 + - openssl >=3.5.2,<4.0a0 + license: MIT + license_family: MIT + size: 666600 + timestamp: 1756834976695 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libnsl-2.0.1-hb9d3cd8_1.conda + sha256: 927fe72b054277cde6cb82597d0fcf6baf127dcbce2e0a9d8925a68f1265eef5 + md5: d864d34357c3b65a4b731f78c0801dc4 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: LGPL-2.1-only + license_family: GPL + size: 33731 + timestamp: 1750274110928 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libntlm-1.8-hb9d3cd8_0.conda + sha256: 3b3f19ced060013c2dd99d9d46403be6d319d4601814c772a3472fe2955612b0 + md5: 7c7927b404672409d9917d49bff5f2d6 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: LGPL-2.1-or-later + size: 33418 + timestamp: 1734670021371 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libogg-1.3.5-hd0c01bc_1.conda + sha256: ffb066ddf2e76953f92e06677021c73c85536098f1c21fcd15360dbc859e22e4 + md5: 68e52064ed3897463c0e958ab5c8f91b + depends: + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + license: BSD-3-Clause + license_family: BSD + size: 218500 + timestamp: 1745825989535 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libopenblas-0.3.30-pthreads_h94d23a6_3.conda + sha256: 200899e5acc01fa29550d2782258d9cf33e55ce4cbce8faed9c6fe0b774852aa + md5: ac2e4832427d6b159576e8a68305c722 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libgfortran + - libgfortran5 >=14.3.0 + constrains: + - openblas >=0.3.30,<0.3.31.0a0 + license: BSD-3-Clause + license_family: BSD + size: 5918287 + timestamp: 1761748180250 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libopenblas-ilp64-0.3.30-pthreads_h3e26593_3.conda + sha256: b78a0a15bfde4217872efb81ccf4d339eaf460e9633f2e6561de4fff386ff41e + md5: b5d918a3a17a7ab24344ab94b604ab89 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libgfortran + - libgfortran5 >=14.3.0 + constrains: + - openblas-ilp64 >=0.3.30,<0.3.31.0a0 + license: BSD-3-Clause + license_family: BSD + size: 5808434 + timestamp: 1761748205098 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libopengl-1.7.0-ha4b6fd6_2.conda + sha256: 215086c108d80349e96051ad14131b751d17af3ed2cb5a34edd62fa89bfe8ead + md5: 7df50d44d4a14d6c31a2c54f2cd92157 + depends: + - __glibc >=2.17,<3.0.a0 + - libglvnd 1.7.0 ha4b6fd6_2 + license: LicenseRef-libglvnd + size: 50757 + timestamp: 1731330993524 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libopenlibm4-0.8.1-hd590300_1.conda + sha256: d176e3b79e723a4102861c06cdc13d60b08210233c42edd6e49e770aece03b3b + md5: e6af610e01d04927a5060c95ce4e0875 + depends: + - libgcc-ng >=12 + license: MIT AND ISC AND BSD-2-Clause + size: 104768 + timestamp: 1698855372633 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libopenssl-static-3.6.0-hb03c661_0.conda + sha256: fb8080d680c6098caefbadbd39b37d65b7e26e41672c35c8e8e0f654188cd968 + md5: 7627b1ada25e7c8ae8f80862e2997f52 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - openssl 3.6.0 h26f9b46_0 + license: Apache-2.0 + license_family: Apache + size: 2878960 + timestamp: 1762839209250 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libopus-1.5.2-hd0c01bc_0.conda + sha256: 786d43678d6d1dc5f88a6bad2d02830cfd5a0184e84a8caa45694049f0e3ea5f + md5: b64523fb87ac6f87f0790f324ad43046 + depends: + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + license: BSD-3-Clause + license_family: BSD + size: 312472 + timestamp: 1744330953241 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libparu-1.0.0-hc6afc67_7100101.conda + sha256: 50144e87b95d1309d2043aa5bf02035b948b1ae9ec6ec44ee97b7aec1cccd70a + md5: fd1d3e26c1b12c70f7449369ae3d9c1a + depends: + - libgcc >=13 + - libstdcxx >=13 + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + - _openmp_mutex >=4.5 + - libsuitesparseconfig >=7.10.1,<8.0a0 + - libblas >=3.9.0,<4.0a0 + - libumfpack >=6.3.5,<7.0a0 + license: GPL-3.0-or-later + license_family: GPL + size: 89738 + timestamp: 1741963824815 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libpciaccess-0.18-hb9d3cd8_0.conda + sha256: 0bd91de9b447a2991e666f284ae8c722ffb1d84acb594dbd0c031bd656fa32b2 + md5: 70e3400cbbfa03e96dcde7fc13e38c7b + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: MIT + license_family: MIT + size: 28424 + timestamp: 1749901812541 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libpng-1.6.50-h421ea60_1.conda + sha256: e75a2723000ce3a4b9fd9b9b9ce77553556c93e475a4657db6ed01abc02ea347 + md5: 7af8e91b0deb5f8e25d1a595dea79614 + depends: + - libgcc >=14 + - __glibc >=2.17,<3.0.a0 + - libzlib >=1.3.1,<2.0a0 + license: zlib-acknowledgement + size: 317390 + timestamp: 1753879899951 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libpq-17.6-h3675c94_2.conda + sha256: 8a078b33f65c5521ae1ed9545ea21efdc46630ff618cdd01aa6a3e698149b27d + md5: e2c2f4c4c20a449b3b4a218797bd7c03 + depends: + - __glibc >=2.17,<3.0.a0 + - icu >=75.1,<76.0a0 + - krb5 >=1.21.3,<1.22.0a0 + - libgcc >=14 + - openldap >=2.6.10,<2.7.0a0 + - openssl >=3.5.2,<4.0a0 + license: PostgreSQL + size: 2726071 + timestamp: 1757976008927 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/librbio-4.3.4-hf02c80a_7100101.conda + sha256: c502b4203cc0d38f49005994b5c80c89660bcd40ff170c529cda90827ec6b1f4 + md5: 4b3a3d711d1c1f76f7f440e51458f512 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libsuitesparseconfig >=7.10.1,<8.0a0 + license: GPL-2.0-or-later + license_family: GPL + size: 46633 + timestamp: 1741963824815 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libsanitizer-15.2.0-hb13aed2_7.conda + sha256: 4d15a66e136fba55bc0e83583de603f46e972f3486e2689628dfd9729a5c3d78 + md5: 4ea6053660330c1bbd4635b945f7626d + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=15.2.0 + - libstdcxx >=15.2.0 + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 5133768 + timestamp: 1759968130105 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libsndfile-1.2.2-hc60ed4a_1.conda + sha256: f709cbede3d4f3aee4e2f8d60bd9e256057f410bd60b8964cb8cf82ec1457573 + md5: ef1910918dd895516a769ed36b5b3a4e + depends: + - lame >=3.100,<3.101.0a0 + - libflac >=1.4.3,<1.5.0a0 + - libgcc-ng >=12 + - libogg >=1.3.4,<1.4.0a0 + - libopus >=1.3.1,<2.0a0 + - libstdcxx-ng >=12 + - libvorbis >=1.3.7,<1.4.0a0 + - mpg123 >=1.32.1,<1.33.0a0 + license: LGPL-2.1-or-later + license_family: LGPL + size: 354372 + timestamp: 1695747735668 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libsolv-0.7.35-h9463b59_0.conda + sha256: 2fc2cdc8ea4dfd9277ae910fa3cfbf342d7890837a2002cf427fd306a869150b + md5: 21769ce326958ec230cdcbd0f2ad97eb + depends: + - libstdcxx >=14 + - libgcc >=14 + - __glibc >=2.17,<3.0.a0 + - libzlib >=1.3.1,<2.0a0 + license: BSD-3-Clause + license_family: BSD + size: 518374 + timestamp: 1754325691186 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libspex-3.2.3-h9226d62_7100101.conda + sha256: 24dffff614943c547ba094f8eb03b412a18cc4654663202f1aab9158bfa875ba + md5: 63323b258079a75133ccecbb0902614d + depends: + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + - _openmp_mutex >=4.5 + - libamd >=3.3.3,<4.0a0 + - libcolamd >=3.3.4,<4.0a0 + - libsuitesparseconfig >=7.10.1,<8.0a0 + - gmp >=6.3.0,<7.0a0 + - mpfr >=4.2.1,<5.0a0 + license: LGPL-2.0-or-later + license_family: LGPL + size: 79220 + timestamp: 1741963824815 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libspqr-4.3.4-h23b7119_7100101.conda + sha256: 52851575496122f9088c9f5a4283da7fbb277d9a877b5ce60a939554df542f3c + md5: c1ee33a71065c1f0efd9c8174d5f18b0 + depends: + - libgcc >=13 + - libstdcxx >=13 + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + - libcholmod >=5.3.1,<6.0a0 + - libblas >=3.9.0,<4.0a0 + - liblapack >=3.9.0,<4.0a0 + - libsuitesparseconfig >=7.10.1,<8.0a0 + license: GPL-2.0-or-later + license_family: GPL + size: 203419 + timestamp: 1741963824816 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libsqlite-3.51.0-hee844dc_0.conda + sha256: 4c992dcd0e34b68f843e75406f7f303b1b97c248d18f3c7c330bdc0bc26ae0b3 + md5: 729a572a3ebb8c43933b30edcc628ceb + depends: + - __glibc >=2.17,<3.0.a0 + - icu >=75.1,<76.0a0 + - libgcc >=14 + - libzlib >=1.3.1,<2.0a0 + license: blessing + size: 945576 + timestamp: 1762299687230 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libssh2-1.11.1-hcf80075_0.conda + sha256: fa39bfd69228a13e553bd24601332b7cfeb30ca11a3ca50bb028108fe90a7661 + md5: eecce068c7e4eddeb169591baac20ac4 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libzlib >=1.3.1,<2.0a0 + - openssl >=3.5.0,<4.0a0 + license: BSD-3-Clause + license_family: BSD + size: 304790 + timestamp: 1745608545575 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libstdcxx-15.2.0-h8f9b012_7.conda + sha256: 1b981647d9775e1cdeb2fab0a4dd9cd75a6b0de2963f6c3953dbd712f78334b3 + md5: 5b767048b1b3ee9a954b06f4084f93dc + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc 15.2.0 h767d61c_7 + constrains: + - libstdcxx-ng ==15.2.0=*_7 + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 3898269 + timestamp: 1759968103436 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/libstdcxx-devel_linux-64-15.2.0-h73f6952_107.conda + sha256: ae5f609b3df4f4c3de81379958898cae2d9fc5d633518747c01d148605525146 + md5: a888a479d58f814ee9355524cc94edf3 + depends: + - __unix + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 13677243 + timestamp: 1759967967095 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libstdcxx-ng-15.2.0-h4852527_7.conda + sha256: 024fd46ac3ea8032a5ec3ea7b91c4c235701a8bf0e6520fe5e6539992a6bd05f + md5: f627678cf829bd70bccf141a19c3ad3e + depends: + - libstdcxx 15.2.0 h8f9b012_7 + license: GPL-3.0-only WITH GCC-exception-3.1 + license_family: GPL + size: 29343 + timestamp: 1759968157195 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libsuitesparseconfig-7.10.1-h901830b_7100101.conda + sha256: d8f32a0b0ee17fbace7af4bd34ad554cc855b9c18e0aeccf8395e1478c161f37 + md5: 57ae1dd979da7aa88a9b38bfa2e1d6b2 + depends: + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + - libgfortran5 >=13.3.0 + - libgfortran + - libgcc >=13 + - _openmp_mutex >=4.5 + license: BSD-3-Clause + license_family: BSD + size: 42708 + timestamp: 1741963824815 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libsystemd0-257.10-hd0affe5_2.conda + sha256: b30c06f60f03c2cf101afeb3452f48f12a2553b4cb631c9460c8a8ccf0813ae5 + md5: b04e0a2163a72588a40cde1afd6f2d18 + depends: + - __glibc >=2.17,<3.0.a0 + - libcap >=2.77,<2.78.0a0 + - libgcc >=14 + license: LGPL-2.1-or-later + size: 491211 + timestamp: 1763011323224 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libtiff-4.7.1-h9d88235_1.conda + sha256: e5f8c38625aa6d567809733ae04bb71c161a42e44a9fa8227abe61fa5c60ebe0 + md5: cd5a90476766d53e901500df9215e927 + depends: + - __glibc >=2.17,<3.0.a0 + - lerc >=4.0.0,<5.0a0 + - libdeflate >=1.25,<1.26.0a0 + - libgcc >=14 + - libjpeg-turbo >=3.1.0,<4.0a0 + - liblzma >=5.8.1,<6.0a0 + - libstdcxx >=14 + - libwebp-base >=1.6.0,<2.0a0 + - libzlib >=1.3.1,<2.0a0 + - zstd >=1.5.7,<1.6.0a0 + license: HPND + size: 435273 + timestamp: 1762022005702 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libtool-2.5.4-h5888daf_0.conda + sha256: c8245c70ba5b075e0cd61f430afbda00b60931603ed4ea31ce89e7fe930e4e3d + md5: 90697d80c181414aa3472199e136a04e + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libltdl 2.4.3a h5888daf_0 + license: GPL-2.0-or-later + license_family: GPL + size: 415044 + timestamp: 1740593851157 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libumfpack-6.3.5-h873dde6_7100101.conda + sha256: 9a2c0049210c0223084c29b39404ad6da6538e7a4d1ed74ee8423212998fd686 + md5: 9626fc7667bc6c901c7a0a4004938c71 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libsuitesparseconfig >=7.10.1,<8.0a0 + - libcholmod >=5.3.1,<6.0a0 + - libblas >=3.9.0,<4.0a0 + - libamd >=3.3.3,<4.0a0 + license: GPL-2.0-or-later + license_family: GPL + size: 404065 + timestamp: 1741963824815 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libunistring-0.9.10-h7f98852_0.tar.bz2 + sha256: e88c45505921db29c08df3439ddb7f771bbff35f95e7d3103bf365d5d6ce2a6d + md5: 7245a044b4a1980ed83196176b78b73a + depends: + - libgcc-ng >=9.3.0 + license: GPL-3.0-only OR LGPL-3.0-only + size: 1433436 + timestamp: 1626955018689 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libunwind-1.6.2-h9c3ff4c_0.tar.bz2 + sha256: f2ac872920833960e514ce9efd8f7c08ce66dd870738d73839d1bce1ac497de6 + md5: a730b2badd586580c5752cc73842e068 + depends: + - libgcc-ng >=9.4.0 + - libstdcxx-ng >=9.4.0 + license: MIT + license_family: MIT + size: 75491 + timestamp: 1638450786937 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libutf8proc-2.11.0-hb04c3b8_0.conda + sha256: f8977233dc19cb8530f3bc71db87124695db076e077db429c3231acfa980c4ac + md5: 34fb73fd2d5a613d8f17ce2eaa15a8a5 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + license: MIT + license_family: MIT + size: 85741 + timestamp: 1757742873826 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libuuid-2.41.2-he9a06e4_0.conda + sha256: e5ec6d2ad7eef538ddcb9ea62ad4346fde70a4736342c4ad87bd713641eb9808 + md5: 80c07c68d2f6870250959dcc95b209d1 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + license: BSD-3-Clause + license_family: BSD + size: 37135 + timestamp: 1758626800002 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libvorbis-1.3.7-h54a6638_2.conda + sha256: ca494c99c7e5ecc1b4cd2f72b5584cef3d4ce631d23511184411abcbb90a21a5 + md5: b4ecbefe517ed0157c37f8182768271c + depends: + - libogg + - libgcc >=14 + - __glibc >=2.17,<3.0.a0 + - libstdcxx >=14 + - libgcc >=14 + - libogg >=1.3.5,<1.4.0a0 + license: BSD-3-Clause + license_family: BSD + size: 285894 + timestamp: 1753879378005 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libwebp-base-1.6.0-hd42ef1d_0.conda + sha256: 3aed21ab28eddffdaf7f804f49be7a7d701e8f0e46c856d801270b470820a37b + md5: aea31d2e5b1091feca96fcfe945c3cf9 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + constrains: + - libwebp 1.6.0 + license: BSD-3-Clause + license_family: BSD + size: 429011 + timestamp: 1752159441324 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libxcb-1.17.0-h8a09558_0.conda + sha256: 666c0c431b23c6cec6e492840b176dde533d48b7e6fb8883f5071223433776aa + md5: 92ed62436b625154323d40d5f2f11dd7 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - pthread-stubs + - xorg-libxau >=1.0.11,<2.0a0 + - xorg-libxdmcp + license: MIT + license_family: MIT + size: 395888 + timestamp: 1727278577118 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libxcrypt-4.4.36-hd590300_1.conda + sha256: 6ae68e0b86423ef188196fff6207ed0c8195dd84273cb5623b85aa08033a410c + md5: 5aa797f8787fe7a17d1b0821485b5adc + depends: + - libgcc-ng >=12 + license: LGPL-2.1-or-later + size: 100393 + timestamp: 1702724383534 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libxkbcommon-1.11.0-he8b52b9_0.conda + sha256: 23f47e86cc1386e7f815fa9662ccedae151471862e971ea511c5c886aa723a54 + md5: 74e91c36d0eef3557915c68b6c2bef96 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - libxcb >=1.17.0,<2.0a0 + - libxml2 >=2.13.8,<2.14.0a0 + - xkeyboard-config + - xorg-libxau >=1.0.12,<2.0a0 + license: MIT/X11 Derivative + license_family: MIT + size: 791328 + timestamp: 1754703902365 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libxml2-2.13.9-h04c0eec_0.conda + sha256: 5d12e993894cb8e9f209e2e6bef9c90fa2b7a339a1f2ab133014b71db81f5d88 + md5: 35eeb0a2add53b1e50218ed230fa6a02 + depends: + - __glibc >=2.17,<3.0.a0 + - icu >=75.1,<76.0a0 + - libgcc >=14 + - libiconv >=1.18,<2.0a0 + - liblzma >=5.8.1,<6.0a0 + - libzlib >=1.3.1,<2.0a0 + license: MIT + license_family: MIT + size: 697033 + timestamp: 1761766011241 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libxslt-1.1.43-h7a3aeb2_0.conda + sha256: 35ddfc0335a18677dd70995fa99b8f594da3beb05c11289c87b6de5b930b47a3 + md5: 31059dc620fa57d787e3899ed0421e6d + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libxml2 >=2.13.8,<2.14.0a0 + license: MIT + license_family: MIT + size: 244399 + timestamp: 1753273455036 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/libzlib-1.3.1-hb9d3cd8_2.conda + sha256: d4bfe88d7cb447768e31650f06257995601f89076080e76df55e3112d4e47dc4 + md5: edb0dca6bc32e4f4789199455a1dbeb8 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + constrains: + - zlib 1.3.1 *_2 + license: Zlib + license_family: Other + size: 60963 + timestamp: 1727963148474 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lp_solve-5.5.2.11-hd590300_0.conda + sha256: 3c28e1494834f77cbd23467fdac0f701cb9a0c17a4251e79e83f0946e788abf9 + md5: c9ca19b1a4b4f1038b23fb8b5918929c + depends: + - libgcc-ng >=12 + license: LGPL-2.1-only + size: 410024 + timestamp: 1697545354266 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lxml-6.0.2-py311hc53b721_0.conda + sha256: 431db76b7d9ecaf1d8689f55f7d9651046abc9aa1f05d0e3d3ccd254cc5c340f + md5: 78a3ed9edec407843eeaad7d6786fdfb + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libxml2 >=2.13.8,<2.14.0a0 + - libxslt >=1.1.43,<2.0a0 + - libzlib >=1.3.1,<2.0a0 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: BSD-3-Clause and MIT-CMU + size: 1600897 + timestamp: 1758535446426 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lz4-c-1.10.0-h5888daf_1.conda + sha256: 47326f811392a5fd3055f0f773036c392d26fdb32e4d8e7a8197eed951489346 + md5: 9de5350a85c4a20c685259b889aa6393 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libstdcxx >=13 + license: BSD-2-Clause + license_family: BSD + size: 167055 + timestamp: 1733741040117 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/lzo-2.10-h280c20c_1002.conda + sha256: 5c6bbeec116e29f08e3dad3d0524e9bc5527098e12fc432c0e5ca53ea16337d4 + md5: 45161d96307e3a447cc3eb5896cf6f8c + depends: + - libgcc >=14 + - __glibc >=2.17,<3.0.a0 + license: GPL-2.0-or-later + license_family: GPL + size: 191060 + timestamp: 1753889274283 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/macse-2.07-hdfd78af_0.tar.bz2 + sha256: c68ac215cc94826c075c71ea8d9b00acc4bfed0248b95d49cc7edc68b59429e2 + md5: 4c0df07d9159231178e59e0b1961dd11 + depends: + - openjdk >=1.5 + license: CeCILL 2.1 + size: 390190 + timestamp: 1679388076563 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/mafft-7.526-h4bc722e_0.conda + sha256: e307b4d817c5f1d1173896d8d54b1b11761b0fc7889bf618ff59dd6407b11d36 + md5: cf3b2acf649bc31e576ed43a2fa4aaa3 + depends: + - __glibc >=2.17 + - __glibc >=2.17,<3.0.a0 + - gawk + - libgcc-ng >=12 + license: BSD-3-Clause + license_family: BSD + size: 2610460 + timestamp: 1720680384725 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/make-4.4.1-hb9d3cd8_2.conda + sha256: d652c7bd4d3b6f82b0f6d063b0d8df6f54cc47531092d7ff008e780f3261bdda + md5: 33405d2a66b1411db9f7242c8b97c9e7 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: GPL-3.0-or-later + license_family: GPL + size: 513088 + timestamp: 1727801714848 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/markdown-it-py-4.0.0-pyhd8ed1ab_0.conda + sha256: 7b1da4b5c40385791dbc3cc85ceea9fad5da680a27d5d3cb8bfaa185e304a89e + md5: 5b5203189eb668f042ac2b0826244964 + depends: + - mdurl >=0.1,<1 + - python >=3.10 + license: MIT + license_family: MIT + size: 64736 + timestamp: 1754951288511 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/matplotlib-3.10.8-py311h38be061_0.conda + sha256: ead3fed3b8709abaf25ac8995ff748ecbbdbfe0f097181754e542ec9dda680c9 + md5: 08b5a4eac150c688c9f924bcb3317e02 + depends: + - matplotlib-base >=3.10.8,<3.10.9.0a0 + - pyside6 >=6.7.2 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + - tornado >=5 + license: PSF-2.0 + license_family: PSF + size: 17484 + timestamp: 1763055534609 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/matplotlib-base-3.10.8-py311h0f3be63_0.conda + sha256: 300bbdb9c90cc1332cb72bc79baf25fa58fd78e0c16f4698ad719b206e42ee1b + md5: 21a0139015232dc0edbf6c2179b5ec24 + depends: + - __glibc >=2.17,<3.0.a0 + - contourpy >=1.0.1 + - cycler >=0.10 + - fonttools >=4.22.0 + - freetype + - kiwisolver >=1.3.1 + - libfreetype >=2.14.1 + - libfreetype6 >=2.14.1 + - libgcc >=14 + - libstdcxx >=14 + - numpy >=1.23 + - numpy >=1.23,<3 + - packaging >=20.0 + - pillow >=8 + - pyparsing >=2.3.1 + - python >=3.11,<3.12.0a0 + - python-dateutil >=2.7 + - python_abi 3.11.* *_cp311 + - qhull >=2020.2,<2020.3.0a0 + - tk >=8.6.13,<8.7.0a0 + license: PSF-2.0 + license_family: PSF + size: 8298261 + timestamp: 1763055503500 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/maven-3.9.11-ha770c72_0.conda + sha256: 69129d0f8a9941fd2df1558a6eef624b9109d8198a6b0ec15152bc4ede7e17f7 + md5: f0e42f3f305ba9165e3c60152bd6c2fe + depends: + - openjdk + license: Apache-2.0 + license_family: APACHE + size: 8866827 + timestamp: 1754929400378 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/mcl-22.282-pl5321h7b50bb2_4.tar.bz2 + sha256: c91e9f49d27199ac595c001749ad166818ddbbb81c682d50d598f4811bbd921e + md5: 5020f96bb41195d95ad24665ddb6cb35 + depends: + - libgcc >=13 + - perl >=5.32.1,<5.33.0a0 *_perl5 + license: GPL-3.0-only + license_family: GPL + size: 2146709 + timestamp: 1748469942874 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/mdurl-0.1.2-pyhd8ed1ab_1.conda + sha256: 78c1bbe1723449c52b7a9df1af2ee5f005209f67e40b6e1d3c7619127c43b1c7 + md5: 592132998493b3ff25fd7479396e8351 + depends: + - python >=3.9 + license: MIT + license_family: MIT + size: 14465 + timestamp: 1733255681319 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/menuinst-2.4.1-py311h38be061_0.conda + sha256: c16073344e6cd1cfb1351b2b7c41388817150f78908a48c55b4a612842dc87dd + md5: 269046366b4c74ebf56aa149e21cbff2 + depends: + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: BSD-3-Clause AND MIT + size: 185521 + timestamp: 1761299901921 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/metaeuk-7.bba0d80-pl5321hd6d6fdc_2.tar.bz2 + sha256: 868647bb8b9a2827cdf98ba80310900b6b7d20fee438b59a1a68616db0452781 + md5: 92d44566468d6b5f4c854d5966069de4 + depends: + - _openmp_mutex >=4.5 + - bzip2 >=1.0.8,<2.0a0 + - gawk + - libgcc >=13 + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + - perl >=5.32.1,<5.33.0a0 *_perl5 + - wget + - zlib + license: GPL-3 + license_family: GPL + size: 4534632 + timestamp: 1733936657790 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/metis-5.1.0-hd0bcaf9_1007.conda + sha256: e8a00971e6d00bd49f375c5d8d005b37a9abba0b1768533aed0f90a422bf5cc7 + md5: 28eb714416de4eb83e2cbc47e99a1b45 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libstdcxx >=13 + license: Apache-2.0 + license_family: APACHE + size: 3923560 + timestamp: 1728064567817 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/miniprot-0.18-h577a1d6_0.tar.bz2 + sha256: a7785a0db01773819ba1c3b897d2987a5ec18f4c80165096c87d924c7e4e845b + md5: e60a264d0b4ba975ac1fb1c8e66753a3 + depends: + - libgcc >=13 + - libzlib >=1.3.1,<2.0a0 + license: MIT + license_family: MIT + size: 70915 + timestamp: 1752254293809 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/mmseqs2-18.8cc5c-hd6d6fdc_0.tar.bz2 + sha256: 466c7bc19eecc58a6043e73f7c80865fc610b4f545af8498565547d87c45d9a9 + md5: a11c5e78c9e77099aa3a2291710292e3 + depends: + - _openmp_mutex >=4.5 + - aria2 + - bzip2 >=1.0.8,<2.0a0 + - gawk + - libgcc >=13 + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + - zlib + license: MIT + size: 133026020 + timestamp: 1753615341810 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/modeltest-ng-0.1.7-hf316886_3.tar.bz2 + sha256: e94c4f49abc8451d053a00fcb963b3b905c47d73be3f6b0e7a97801b11de5763 + md5: 36086d1693af82705be7a39d497707e4 + depends: + - libgcc >=13 + - libstdcxx >=13 + - openmpi >=4.1.6,<5.0a0 + license: GPL-3.0 + license_family: GPL + size: 6047364 + timestamp: 1734158893881 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/mpfr-4.2.1-h90cbb55_3.conda + sha256: f25d2474dd557ca66c6231c8f5ace5af312efde1ba8290a6ea5e1732a4e669c0 + md5: 2eeb50cab6652538eee8fc0bc3340c81 + depends: + - __glibc >=2.17,<3.0.a0 + - gmp >=6.3.0,<7.0a0 + - libgcc >=13 + license: LGPL-3.0-only + license_family: LGPL + size: 634751 + timestamp: 1725746740014 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/mpg123-1.32.9-hc50e24c_0.conda + sha256: 39c4700fb3fbe403a77d8cc27352fa72ba744db487559d5d44bf8411bb4ea200 + md5: c7f302fd11eeb0987a6a5e1f3aed6a21 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libstdcxx >=13 + license: LGPL-2.1-only + license_family: LGPL + size: 491140 + timestamp: 1730581373280 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/mpi-1.0-openmpi.tar.bz2 + sha256: 54cf44ee2c122bce206f834a825af06e3b14fc4fd58c968ae9329715cc281d1e + md5: 1dcc49e16749ff79ba2194fa5d4ca5e7 + license: BSD 3-clause + size: 4204 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/mrbayes-3.2.7-hd0d793b_7.tar.bz2 + sha256: a1463f247a0ee85653f91f65d2a15f55aa74e4af429b2b84356dd124c25d08cb + md5: c6a11f09208623646e4114e19219ecfe + depends: + - beagle-lib <4 + - libgcc >=13 + - ncurses >=6.5,<7.0a0 + - openmpi >=4.1.6,<5.0a0 + - readline >=8.2,<9.0a0 + license: GPLv3 + license_family: GPL + size: 6193186 + timestamp: 1734212205994 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/munkres-1.1.4-pyhd8ed1ab_1.conda + sha256: d09c47c2cf456de5c09fa66d2c3c5035aa1fa228a1983a433c47b876aa16ce90 + md5: 37293a85a0f4f77bbd9cf7aaefc62609 + depends: + - python >=3.9 + license: Apache-2.0 + license_family: Apache + size: 15851 + timestamp: 1749895533014 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/muscle-3.8.1551-h9948957_9.tar.bz2 + sha256: 07f12b158989bd98b4d7af97a2739ccd7bbef0728ddc5a036bd9784363ab0ba9 + md5: abeab9c9f2802eae4f867260650837f6 + depends: + - libgcc >=13 + - libstdcxx >=13 + license: GPL-3.0-only + size: 273543 + timestamp: 1752497842816 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/mysql-connector-c-6.1.11-h659d440_1008.conda + sha256: 0d796529e90bf6f511fd7199acfb5cc788e04d853e1afd48aa6bad33a6e49008 + md5: 149e0b89cbbe09397f1147cb3736bcbe + depends: + - libgcc-ng >=12 + - libstdcxx-ng >=12 + - openssl >=3.1.3,<4.0a0 + license: GPL-2.0-only + license_family: GPL + size: 1268505 + timestamp: 1697651492989 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/ncbi-vdb-3.2.1-h9948957_0.tar.bz2 + sha256: f42b1398b178b6de64a9d180ef29bac9192b5fcd9492eced4eca854c22f6aba9 + md5: c80d2359e66e4dd94bceedd55b752dff + depends: + - libgcc >=13 + - libstdcxx >=13 + license: Public Domain + size: 11067140 + timestamp: 1742339687061 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/ncurses-6.5-h2d0b736_3.conda + sha256: 3fde293232fa3fca98635e1167de6b7c7fda83caf24b9d6c91ec9eefb4f4d586 + md5: 47e340acb35de30501a76c7c799c41d7 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: X11 AND BSD-3-Clause + size: 891641 + timestamp: 1738195959188 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/nlohmann_json-abi-3.12.0-h0f90c79_1.conda + sha256: 2a909594ca78843258e4bda36e43d165cda844743329838a29402823c8f20dec + md5: 59659d0213082bc13be8500bab80c002 + license: MIT + license_family: MIT + size: 4335 + timestamp: 1758194464430 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/nspr-4.38-h29cc59b_0.conda + sha256: e3664264bd936c357523b55c71ed5a30263c6ba278d726a75b1eb112e6fb0b64 + md5: e235d5566c9cc8970eb2798dd4ecf62f + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + license: MPL-2.0 + license_family: MOZILLA + size: 228588 + timestamp: 1762348634537 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/nss-3.117-h445c969_0.conda + sha256: 85f2d6d93199454818866b355834a8c5dc64a87e14da3b242208c9dc2156852a + md5: 970af0bfac9644ddbf7e91c1336b231b + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libsqlite >=3.50.4,<4.0a0 + - libstdcxx >=14 + - libzlib >=1.3.1,<2.0a0 + - nspr >=4.37,<5.0a0 + license: MPL-2.0 + license_family: MOZILLA + size: 2045760 + timestamp: 1759509411326 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/numpy-2.3.4-py311h2e04523_0.conda + sha256: 67cc072b8f5c157df4228a1a2291628e5ca2360f48ef572a64e2cf2bf55d2e25 + md5: d84afde5a6f028204f24180ff87cf429 + depends: + - python + - libstdcxx >=14 + - libgcc >=14 + - __glibc >=2.17,<3.0.a0 + - python_abi 3.11.* *_cp311 + - liblapack >=3.9.0,<4.0a0 + - libcblas >=3.9.0,<4.0a0 + - libblas >=3.9.0,<4.0a0 + constrains: + - numpy-base <0a0 + license: BSD-3-Clause + license_family: BSD + size: 9418119 + timestamp: 1761162089374 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/oniguruma-6.9.10-hb9d3cd8_0.conda + sha256: bbff8a60f70d5ebab138b564554f28258472e1e63178614562d4feee29d10da2 + md5: 6ce853cb231f18576d2db5c2d4cb473e + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: BSD-2-Clause + license_family: BSD + size: 248670 + timestamp: 1735727084819 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openblas-ilp64-0.3.30-pthreads_h3d04fff_3.conda + sha256: d45ea7968fbc6bce123b453d599ef5c955330b3b193e4b30b267182a6c0106b7 + md5: 3b94d5cdb59f2ea990f15c667550f689 + depends: + - libopenblas-ilp64 0.3.30 pthreads_h3e26593_3 + license: BSD-3-Clause + license_family: BSD + size: 5927114 + timestamp: 1761748219065 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openjdk-25.0.1-h5755bd7_0.conda + sha256: 19b2268bf2d1fc4b4f48a68b9bfac620370c1b7f539671279053b0d3bcc348f1 + md5: a40ce38da029d1d272bfd9bd7510f901 + depends: + - __glibc >=2.17,<3.0.a0 + - alsa-lib >=1.2.14,<1.3.0a0 + - fontconfig >=2.15.0,<3.0a0 + - fonts-conda-ecosystem + - giflib >=5.2.2,<5.3.0a0 + - harfbuzz >=12.1.0 + - lcms2 >=2.17,<3.0a0 + - libcups >=2.3.3,<2.4.0a0 + - libfreetype >=2.14.1 + - libfreetype6 >=2.14.1 + - libgcc >=14 + - libjpeg-turbo >=3.1.0,<4.0a0 + - libpng >=1.6.50,<1.7.0a0 + - libstdcxx >=14 + - libzlib >=1.3.1,<2.0a0 + - xorg-libx11 >=1.8.12,<2.0a0 + - xorg-libxext >=1.3.6,<2.0a0 + - xorg-libxi >=1.8.2,<2.0a0 + - xorg-libxrandr >=1.5.4,<2.0a0 + - xorg-libxrender >=0.9.12,<0.10.0a0 + - xorg-libxt >=1.3.1,<2.0a0 + - xorg-libxtst >=1.2.5,<2.0a0 + license: GPL-2.0-or-later WITH Classpath-exception-2.0 + license_family: GPL + size: 117033638 + timestamp: 1762057253080 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openjpeg-2.5.4-h55fea9a_0.conda + sha256: 3900f9f2dbbf4129cf3ad6acf4e4b6f7101390b53843591c53b00f034343bc4d + md5: 11b3379b191f63139e29c0d19dee24cd + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libpng >=1.6.50,<1.7.0a0 + - libstdcxx >=14 + - libtiff >=4.7.1,<4.8.0a0 + - libzlib >=1.3.1,<2.0a0 + license: BSD-2-Clause + license_family: BSD + size: 355400 + timestamp: 1758489294972 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openldap-2.6.10-he970967_0.conda + sha256: cb0b07db15e303e6f0a19646807715d28f1264c6350309a559702f4f34f37892 + md5: 2e5bf4f1da39c0b32778561c3c4e5878 + depends: + - __glibc >=2.17,<3.0.a0 + - cyrus-sasl >=2.1.27,<3.0a0 + - krb5 >=1.21.3,<1.22.0a0 + - libgcc >=13 + - libstdcxx >=13 + - openssl >=3.5.0,<4.0a0 + license: OLDAP-2.8 + license_family: BSD + size: 780253 + timestamp: 1748010165522 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openlibm-0.8.1-hd590300_1.conda + sha256: 689a90096c3fdd76bd3620bd17a8de2924cf9e1ebe7c0d2d52c6f06eda2ad235 + md5: 6eba22eb06d69e53d0ca01eef42bc675 + depends: + - libgcc-ng >=12 + - libopenlibm4 0.8.1 hd590300_1 + license: MIT AND ISC AND BSD-2-Clause + size: 29204 + timestamp: 1698855384579 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openmpi-4.1.6-hc5af2df_101.conda + sha256: f0769dd891e1735be4606ec8643951e5cbca199f774e58c7d933f70a70134ce4 + md5: f9a2ad0088ee38f396350515fa37d243 + depends: + - libgcc-ng >=12 + - libgfortran-ng + - libgfortran5 >=12.3.0 + - libstdcxx-ng >=12 + - libzlib >=1.2.13,<2.0.0a0 + - mpi 1.0 openmpi + - zlib + constrains: + - cudatoolkit >= 10.2 + - ucx >=1.15.0,<2.0a0 + - libpmix ==0.0.0 + - libprrte ==0.0.0 + license: BSD-3-Clause + license_family: BSD + size: 4069632 + timestamp: 1696593196408 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/openssl-3.6.0-h26f9b46_0.conda + sha256: a47271202f4518a484956968335b2521409c8173e123ab381e775c358c67fe6d + md5: 9ee58d5c534af06558933af3c845a780 + depends: + - __glibc >=2.17,<3.0.a0 + - ca-certificates + - libgcc >=14 + license: Apache-2.0 + license_family: Apache + size: 3165399 + timestamp: 1762839186699 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/orthofinder-3.1.0-hdfd78af_1.conda + sha256: 3ab567bd89d7b5d0a7edfc929e21652999951fd02630327ba60e64e4f7e58f3e + md5: 9af33232ce8428bf464962c16c69f8df + depends: + - aster + - biopython + - blast + - bzip2 + - diamond <2.1|>=2.1.7 + - ete3 + - famsa + - fastme + - fasttree + - iqtree + - mafft + - mcl + - mmseqs2 + - muscle <5 + - numpy + - python >=3.8,<3.12 + - raxml + - raxml-ng + - rich + - scikit-learn + - scipy + - six + license: GPL-3.0-only + license_family: GPL3 + size: 1872721 + timestamp: 1756347160836 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/ossuuid-1.6.2-h5888daf_1001.conda + sha256: fec82722c32caf0d4f1f8f4148786dc8898e903b32a2d9974b979f8b597cab12 + md5: 6cc16cf6cd2fd0b22844735f52468a98 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libstdcxx >=13 + license: GPL-2.0-or-later + license_family: GPL + size: 54218 + timestamp: 1736761658961 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/p7zip-16.02-h9c3ff4c_1001.tar.bz2 + sha256: 01aecc8f648ed0825ecf4c384c2f4759146ef8f2815348efddebf42ec769419c + md5: 941066943c0cac69d5aa52189451aa5f + depends: + - libgcc-ng >=9.4.0 + - libstdcxx-ng >=9.4.0 + license: LGPL-2.0-or-later + license_family: LGPL + size: 2304688 + timestamp: 1650994775055 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/packaging-25.0-pyh29332c3_1.conda + sha256: 289861ed0c13a15d7bbb408796af4de72c2fe67e2bcb0de98f4c3fce259d7991 + md5: 58335b26c38bf4a20f399384c33cbcf9 + depends: + - python >=3.8 + - python + license: Apache-2.0 + license_family: APACHE + size: 62477 + timestamp: 1745345660407 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/pal2nal-14.1-pl5321hdfd78af_3.tar.bz2 + sha256: aff933ffefe2ecbc31b6e38bbbb2a19b5b34c2942b3d2078d7f5848432d494f6 + md5: 2afdf27b9f28038bfb670e2059deca04 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-getopt-long + license: GPLv2.0 + license_family: GPL + size: 22693 + timestamp: 1642293349880 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pandas-2.3.3-py311hed34c8f_1.conda + sha256: c97f796345f5b9756e4404bbb4ee049afd5ea1762be6ee37ce99162cbee3b1d3 + md5: 72e3452bf0ff08132e86de0272f2fbb0 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - numpy >=1.22.4 + - numpy >=1.23,<3 + - python >=3.11,<3.12.0a0 + - python-dateutil >=2.8.2 + - python-tzdata >=2022.7 + - python_abi 3.11.* *_cp311 + - pytz >=2020.1 + constrains: + - beautifulsoup4 >=4.11.2 + - scipy >=1.10.0 + - pytables >=3.8.0 + - gcsfs >=2022.11.0 + - odfpy >=1.4.1 + - xlsxwriter >=3.0.5 + - openpyxl >=3.1.0 + - html5lib >=1.1 + - python-calamine >=0.1.7 + - qtpy >=2.3.0 + - pyxlsb >=1.0.10 + - xarray >=2022.12.0 + - pandas-gbq >=0.19.0 + - numexpr >=2.8.4 + - tzdata >=2022.7 + - pyreadstat >=1.2.0 + - lxml >=4.9.2 + - pyqt5 >=5.15.9 + - s3fs >=2022.11.0 + - fastparquet >=2022.12.0 + - psycopg2 >=2.9.6 + - xlrd >=2.0.1 + - matplotlib >=3.6.3 + - blosc >=1.21.3 + - numba >=0.56.4 + - sqlalchemy >=2.0.0 + - fsspec >=2022.11.0 + - pyarrow >=10.0.1 + - zstandard >=0.19.0 + - bottleneck >=1.3.6 + - tabulate >=0.9.0 + license: BSD-3-Clause + license_family: BSD + size: 15337715 + timestamp: 1759266002530 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pandoc-3.8.2.1-ha770c72_0.conda + sha256: 6b92e15cbc84ce4a0171ca0a9b9f483888a9065b17302d1503c0cacfcf8abd56 + md5: 47432e6a6fb5d9697564185e1907138a + license: GPL-2.0-or-later + license_family: GPL + size: 22018364 + timestamp: 1760964197643 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pango-1.56.4-hadf4263_0.conda + sha256: 3613774ad27e48503a3a6a9d72017087ea70f1426f6e5541dbdb59a3b626eaaf + md5: 79f71230c069a287efe3a8614069ddf1 + depends: + - __glibc >=2.17,<3.0.a0 + - cairo >=1.18.4,<2.0a0 + - fontconfig >=2.15.0,<3.0a0 + - fonts-conda-ecosystem + - fribidi >=1.0.10,<2.0a0 + - harfbuzz >=11.0.1 + - libexpat >=2.7.0,<3.0a0 + - libfreetype >=2.13.3 + - libfreetype6 >=2.13.3 + - libgcc >=13 + - libglib >=2.84.2,<3.0a0 + - libpng >=1.6.49,<1.7.0a0 + - libzlib >=1.3.1,<2.0a0 + license: LGPL-2.1-or-later + size: 455420 + timestamp: 1751292466873 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/pasta-1.9.3-py311hefa8cab_0.tar.bz2 + sha256: ddc495ebfdad715427518b28b07559bdbe78a97160b31049b1d156db08d33853 + md5: 8aef89c53a98822c70a9069ccdb7311e + depends: + - _openmp_mutex >=4.5 + - clustalw >=2.1,<3.0a0 + - dendropy >=5.0.8,<6.0a0 + - fasttree >=2.1.11,<3.0a0 + - hmmer >=3.4,<3.5.0a0 + - libgcc >=13 + - libgomp + - mafft >=7.526,<8.0a0 + - muscle <4 + - muscle >=3.8.1551,<4.0a0 + - openjdk + - openmpi >=4.1.6,<5.0a0 + - pcre >=8.45,<9.0a0 + - prank >=170427,<170428.0a0 + - pymongo + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + - raxml >=8.2.13,<9.0a0 + license: GPL-3.0-or-later + license_family: GPL3 + size: 1289197 + timestamp: 1749151368561 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pcre-8.45-h9c3ff4c_0.tar.bz2 + sha256: 8f35c244b1631a4f31fb1d66ab6e1d9bfac0ca9b679deced1112c7225b3ad138 + md5: c05d1820a6d34ff07aaaab7a9b7eddaa + depends: + - libgcc-ng >=9.3.0 + - libstdcxx-ng >=9.3.0 + license: BSD-3-Clause + license_family: BSD + size: 259377 + timestamp: 1623788789327 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pcre2-10.46-h1321c63_0.conda + sha256: 5c7380c8fd3ad5fc0f8039069a45586aa452cf165264bc5a437ad80397b32934 + md5: 7fa07cb0fb1b625a089ccc01218ee5b1 + depends: + - __glibc >=2.17,<3.0.a0 + - bzip2 >=1.0.8,<2.0a0 + - libgcc >=14 + - libzlib >=1.3.1,<2.0a0 + license: BSD-3-Clause + license_family: BSD + size: 1209177 + timestamp: 1756742976157 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-5.32.1-7_hd590300_perl5.conda + build_number: 7 + sha256: 9ec32b6936b0e37bcb0ed34f22ec3116e75b3c0964f9f50ecea5f58734ed6ce9 + md5: f2cfec9406850991f4e3d960cc9e3321 + depends: + - libgcc-ng >=12 + - libxcrypt >=4.4.36 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 13344463 + timestamp: 1703310653947 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/perl-alien-build-2.84-pl5321h7b50bb2_1.tar.bz2 + sha256: ef28ca71abf0b875953d1d68a1cf14a0d14de7df20b33d24d98d9aeaf0a8c47e + md5: 1966b7077b4cacfa9e2a2cb65723e9c1 + depends: + - libgcc >=13 + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-capture-tiny + - perl-ffi-checklib 0.28.* + - perl-file-chdir + - perl-file-which + - perl-path-tiny + - perl-test2-suite 0.000163.* + license: perl_5 + size: 200097 + timestamp: 1745269972596 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/perl-alien-libxml2-0.17-pl5321h577a1d6_1.tar.bz2 + sha256: 862974100504e2eb4c65eb387406f5cca4940d801a197dc7bce68f6d84f8308a + md5: f89f1fd0808f00264371a41186abbcef + depends: + - libgcc >=13 + - libzlib >=1.3.1,<2.0a0 + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-alien-build >=2.84,<3.0a0 + license: perl_5 + size: 14197 + timestamp: 1735914026197 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-app-cpanminus-1.7048-pl5321hd8ed1ab_0.conda + sha256: 20850458e691b12fc4da209e204fc4c4d1abe93311913929dc536b9b3857b681 + md5: adb8fae79589a7bb03aa9ad976e3bd92 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 229935 + timestamp: 1730267467434 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-archive-tar-3.04-pl5321hdfd78af_0.tar.bz2 + sha256: ff80ba34551d4e051f4eaff48df53a6c63695dcf00e670c89a241a04f0b662ae + md5: 27d0df347c48c48b363651b64fb6fb4c + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-io-compress + - perl-io-zlib + - perl-pathtools + license: Perl_5 + size: 35243 + timestamp: 1749770047223 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-business-isbn-3.007-pl5321hd8ed1ab_0.tar.bz2 + sha256: fa0b99650ac3bc098d4b34d23eea8f5101f7a5e8c04fc50fac6b8bff70161848 + md5: a7a3d7614e1a73b8d9c20030651d6006 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-business-isbn-data >=20191107 + license: Artistic-2.0 + size: 18309 + timestamp: 1665411030544 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-business-isbn-data-20210112.006-pl5321hd8ed1ab_0.tar.bz2 + sha256: 457f7bdea5c7154472a18ab10d7c40144d65793446d9cb40877dc9c37c6b3b59 + md5: a70f08650ec3919ee5e599834f1349d3 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-carp + license: Artistic-2.0 + license_family: OTHER + size: 21540 + timestamp: 1660390931370 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-capture-tiny-0.48-pl5321ha770c72_1.tar.bz2 + sha256: 0c0e8e7a342edcdd8638bc2ecf1fca13917b67bcf40890bcff7a7bf4db2ff4d3 + md5: 4859b3d284090166394875e68184dc9c + depends: + - perl >=5.32.1,<5.33.0a0 *_perl5 + license: Artistic-2.0 + size: 16519 + timestamp: 1643301265873 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-carp-1.50-pl5321hd8ed1ab_0.tar.bz2 + sha256: 1981e31113e1e77a2cdc13db657c636f047cd3be2a64d9a0bffac03c5427c1bd + md5: bdddc03e28019b902da71b722f2288d7 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-exporter + - perl-extutils-makemaker + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 22257 + timestamp: 1636653208008 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-class-method-modifiers-2.13-pl5321ha770c72_0.tar.bz2 + sha256: bed763551ecf5e47249f4596b642f6279d64ee6e1c73998236456829c29ea6c2 + md5: 7b7ec3dd22b056c2d5ea53f9033c26b8 + depends: + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-test-fatal 0.016.* + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 22098 + timestamp: 1666342663240 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-common-sense-3.75-pl5321hd8ed1ab_0.tar.bz2 + sha256: 38ef218e9b9d55b9fbdce6b31cf81bcf6f1b16f21b8e7cb9279b41399522a320 + md5: ef70dc77e8b10bbb62f5e843b401ef0e + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 20291 + timestamp: 1660429950685 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-compress-raw-bzip2-2.214-pl5321hda65f42_0.conda + sha256: 2cbd3fcb7028415654057ab3142cee1e5cc4cf9bdd0e590f072e8ecd29cb6382 + md5: 1ec33b9f74a71c17cc5579e35027832a + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - perl >=5.32.1,<5.33.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 55733 + timestamp: 1761332551110 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-compress-raw-zlib-2.214-pl5321h4dac143_0.conda + sha256: 4fd3a11f8dbd2c9d1a1c5547e344ad20412a3f0d36967502d9764e1b2734dd47 + md5: 943e7033e6f709adf0d89a3af51ab81b + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - perl >=5.32.1,<5.33.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 81114 + timestamp: 1761332700914 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-constant-1.33-pl5321hd8ed1ab_0.tar.bz2 + sha256: cfd5ef9af8de221292f7059d1ff88ff10f1340fc4dbbd0dc81ce2275bf76651d + md5: 7f9fc9cfa08a3fe36ffcff820c65c3ac + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 15711 + timestamp: 1636645107290 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-db_file-1.858-pl5321hb9d3cd8_0.conda + sha256: 9c95150bc621a32406da4297e9cca4c73565f2b01ac16b1ad78d407ae287978c + md5: 2dc3e251150589a85fc4219568b12a25 + depends: + - __glibc >=2.17,<3.0.a0 + - libdb >=6.2.32,<6.3.0a0 + - libgcc >=13 + - perl >=5.32.1,<5.33.0a0 *_perl5 + license: Artistic-1.0-Perl OR GPL-1.0-or-later + size: 63679 + timestamp: 1733156864666 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-dbi-1.647-pl5321hb03c661_0.conda + sha256: b336208402e035a7d92a53c07e4b1f29a84d918b458a9aaec557bfac49793b8e + md5: 313b58e6fc82ad7b32b7c5dd6aa1919a + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - perl >=5.32.1,<5.33.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 601162 + timestamp: 1755720107981 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-encode-3.21-pl5321hb9d3cd8_1.conda + sha256: 4cbe4125efe8763e1ad44448852b04481002f1dac84b6052cec1626df79e3a16 + md5: a418a0e7010007df65768e4e4b96dd14 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-exporter + - perl-parent + - perl-storable + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 1731511 + timestamp: 1728247059262 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-exporter-5.74-pl5321hd8ed1ab_0.tar.bz2 + sha256: 42271d0b79043a10a89044acb5febea50046b745dd2fc37e02943bc3bc75bf8e + md5: fd2eac4e35f8c970870a3961c1df3e29 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 19071 + timestamp: 1636696009075 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-exporter-tiny-1.002002-pl5321hd8ed1ab_0.tar.bz2 + sha256: abdf86828a12a389d0feb0d70501b267842557bae11820e266526aeb6ab2bebe + md5: 48d709826875be1f2c108d3d1d8efec7 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 28592 + timestamp: 1660341479867 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-extutils-makemaker-7.70-pl5321hd8ed1ab_0.conda + sha256: 1d3f342ca74cf2948c3edcfe0d3367b1db0fc64bb163393a2e025336dec3a40c + md5: ec3e57ed34f7765bfc7054a05868ce5d + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 157323 + timestamp: 1679847836884 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-ffi-checklib-0.28-pl5321hdfd78af_0.tar.bz2 + sha256: 4ac8de71ec66b6f13785f134be7878507f0003844e279be6290b7dc72cc2aa88 + md5: 05451eeb5e82ccf46831248c55c6218a + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: perl_5 + size: 16150 + timestamp: 1648502850959 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-file-chdir-0.1011-pl5321hd8ed1ab_0.tar.bz2 + sha256: 736b5c79e0ea19efb8f97f628a508cf42a1d5b6bc5533df1bd454651e28950c2 + md5: c057570093f9298e9c5e4391877f2301 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-pathtools + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 16973 + timestamp: 1660380425665 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-file-path-2.18-pl5321hd8ed1ab_0.tar.bz2 + sha256: ac7a0e0c36ec199709c8523132ded35efb84941a1a476b271fd25362dfbf5339 + md5: e13e456f61b8261ba074c0aa93086afe + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-exporter + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 22382 + timestamp: 1636696260633 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-file-temp-0.2304-pl5321hd8ed1ab_0.tar.bz2 + sha256: 91f9fe152b38441386053876198d0c9437e7b4b5fa7da1e5db48b9e1b21fdae9 + md5: 0a039c4fc36748942287ee10d0431515 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-carp + - perl-constant + - perl-exporter + - perl-file-path + - perl-parent + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 31509 + timestamp: 1636696164992 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-file-which-1.24-pl5321hd8ed1ab_0.tar.bz2 + sha256: 901e9c4177b6bdb23600446fd64a105a437112e63bc23a6163f669bc5098c502 + md5: 94e5b2c9d56ef7c4c70ddf51ac38ae3d + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 17329 + timestamp: 1636695979827 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-getopt-long-2.58-pl5321hdfd78af_0.tar.bz2 + sha256: 94c17b270316f9b38cf1afa0b1724cdb6bd8349207568dccc1156b9014708c7d + md5: 10469a3defbd6dbab17d3d10dd6fd828 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: unknown + size: 33785 + timestamp: 1718113256479 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-importer-0.026-pl5321hd8ed1ab_0.tar.bz2 + sha256: f40d5302de8c81282b5216edd8c65aecc551ef725db3351c208bff9b077e66a2 + md5: 66e17c342d13c39a12c1059f13f9b72c + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 26921 + timestamp: 1660471365476 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-inc-latest-0.500-pl5321ha770c72_0.conda + sha256: ff6b6ffffd6ecbed19a00ee41bb54c283d171f3feff7f45e840441551b5ab876 + md5: 0879971f134628d37bc4b13581113134 + depends: + - perl >=5.32.1,<5.33.0a0 *_perl5 + license: Apache-2.0 + license_family: APACHE + size: 16039 + timestamp: 1669240163980 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/perl-io-compress-2.213-pl5321h503566f_0.conda + sha256: a68e0bcac001000a03ba5e37881812c6ce462886d5fb7f1a19204b72af00be9e + md5: b454321921fd315d492e0126ef45b8ef + depends: + - libgcc >=13 + - libstdcxx >=13 + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-compress-raw-bzip2 >=2.213 + - perl-compress-raw-zlib >=2.213 + - perl-encode + - perl-scalar-list-utils + license: Perl_5 + size: 86935 + timestamp: 1756385532279 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-io-zlib-1.15-pl5321hdfd78af_1.tar.bz2 + sha256: 771a44b338cac68a22893897450222b886e24bbb291b014897d561f2ba3b588f + md5: db92645dabe9467115729e4479841b5d + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 12760 + timestamp: 1752069858135 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-json-4.10-pl5321hdfd78af_1.tar.bz2 + sha256: c30768595793865d67fe2bf76a34e696a0664ae1f1cef34e31cb16618af22d61 + md5: c6c43c11e14d90b836f42c611e106ea9 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-json-xs + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 57728 + timestamp: 1722414214816 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/perl-json-xs-4.04-pl5321h9948957_0.conda + sha256: f16cf35917a5d75fb55cfc69d8651dc72ff4e06370142b447a4864f5de7d97c6 + md5: 7916b2e794393b3389535ba750f489da + depends: + - libgcc >=13 + - libstdcxx >=13 + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-common-sense + - perl-types-serialiser + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 70770 + timestamp: 1757352008878 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-list-moreutils-0.430-pl5321hdfd78af_0.tar.bz2 + sha256: 2190cc8430bb218ea80f5fc5e2bf75e4e20a27fb83e8c3cb789a18c32854a56c + md5: 7f04c79d216d0f8e7b6d5a51de4aafa0 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-exporter-tiny + - perl-list-moreutils-xs >=0.430 + license: apache_2_0 + size: 32468 + timestamp: 1644871004171 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/perl-list-moreutils-xs-0.430-pl5321h7b50bb2_5.tar.bz2 + sha256: fe2d360770fe5b856ee1e625eac6b07cea386c64b0866e323c6535eb62b9ceee + md5: 9192770eb08524038c3fcf24e915b10c + depends: + - libgcc >=13 + - perl >=5.32.1,<5.33.0a0 *_perl5 + license: apache_2_0 + size: 51506 + timestamp: 1741776389435 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-module-build-0.4234-pl5321ha770c72_1.conda + sha256: 40bc64172a969e87ade758320049c86c6a0d83cf8aeaa69ad218fa463ac6cb5b + md5: 358d42e9f08dd917b649089526e2ae1f + depends: + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-inc-latest 0.500.* + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 137087 + timestamp: 1738241035046 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-moo-2.005004-pl5321ha770c72_0.tar.bz2 + sha256: e82a11e50392c965e88db4cba4d23ed0be5c4917319385ea23d32d9484246227 + md5: 3499fef1ed2c74f253b1f032f22deb51 + depends: + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-class-method-modifiers 2.13.* + - perl-role-tiny 2.002004.* + - perl-sub-quote 2.006006.* + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 47669 + timestamp: 1666344231925 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-parallel-forkmanager-2.04-pl5321hdfd78af_0.conda + sha256: d71f83a9d415beb11e4f9832122236cd13686393f6336863ebc8a46d5e3e315f + md5: c0d5b1238870d70b3af64854901d7f62 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-carp + - perl-file-path + - perl-file-temp + - perl-moo + - perl-storable + license: perl_5 + size: 24120 + timestamp: 1756574734791 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-parent-0.243-pl5321hd8ed1ab_0.conda + sha256: ec57d9e56ba86d840d5a9e65c665365fb6a290851cd13e6719628f3295eabf34 + md5: 314caa3b72d65f8078d426c6721dcacc + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 13933 + timestamp: 1733429736293 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-path-tiny-0.124-pl5321hd8ed1ab_0.tar.bz2 + sha256: 03b980520f6bcf7612fdee607e571ff91660d8fbfeb57ae9350921c5a38459d1 + md5: 456a8757a2e5272fb1dcb6435e037f3a + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: Apache-2.0 + license_family: APACHE + size: 41525 + timestamp: 1662587359913 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-pathtools-3.75-pl5321hb9d3cd8_2.conda + sha256: a80bc265aa749ae03fdd6d5f2098312ea6f43cc193ea7aba0e6e4304d84ce8c0 + md5: 03d88c89dfac8a26adcdeff242f94007 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-carp + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 50681 + timestamp: 1741783312134 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-role-tiny-2.002004-pl5321ha770c72_0.tar.bz2 + sha256: 7895ed40c44c08ee3ddbb48a3ee70e1eda32a6c529b632af98663e7603adda08 + md5: 83b5b630aeddca09e26fded9af4f3ce5 + depends: + - perl >=5.32.1,<5.33.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 23283 + timestamp: 1666339833893 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-scalar-list-utils-1.70-pl5321hb03c661_0.conda + sha256: 00678b34a57df53fa31585ff1579f42a6dfc8b1b059fd1b386c6e89573959a9f + md5: 046486011bf7674e3c2f2a1513ac4f3d + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - perl >=5.32.1,<5.33.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 51958 + timestamp: 1753965684679 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-scope-guard-0.21-pl5321hd8ed1ab_0.tar.bz2 + sha256: f0cd98e792fe43fcef81af1848f0d985a87b953043f1f6551d2ffdf15daca62c + md5: daea4e61dfdf06fef9a51dce492508f4 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 16123 + timestamp: 1660472428828 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-storable-3.15-pl5321hb9d3cd8_2.conda + sha256: 25beba40154a394189d0ff2afd31683d79c9106d47e77e12d20900d053dffaf6 + md5: 212f63a5c7c753ecf0a74c349b9270c5 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - perl >=5.32.1,<5.33.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 71475 + timestamp: 1741353849888 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-sub-info-0.002-pl5321hd8ed1ab_0.tar.bz2 + sha256: 0485649e2aaccdbe46b64d35e07d13f885d28f438f41bee54ecd7e9a9d94d472 + md5: 5b48dcf7df9e4e01d45de2c1f494d830 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-carp + - perl-importer + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 19020 + timestamp: 1663840434982 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-sub-quote-2.006006-pl5321ha770c72_0.tar.bz2 + sha256: 4200429318a1a2d9276c4095bb63e29d655fc98208341b6db0752788d3016587 + md5: 3d58e92f33b2382c10ac7eccd883a42f + depends: + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-test-fatal 0.016.* + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 26698 + timestamp: 1666342630627 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-term-table-0.025-pl5321hdfd78af_0.conda + sha256: edb2a2e6d3605362fc1cd403ce88e5e69a211a0d10570a422fdc769e240b5693 + md5: 4740773caf9bdc262d2607d1814f7df0 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-carp + - perl-importer + license: perl_5 + size: 24620 + timestamp: 1756668297689 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-test-fatal-0.016-pl5321ha770c72_0.tar.bz2 + sha256: d5420f90b3bca641cdad0c5674b356a060bf1587e82d3828f382a466bdaba6d1 + md5: 5c397b1fd3095004f4ff149af1c0dd3c + depends: + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-try-tiny 0.31.* + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 19528 + timestamp: 1666339958976 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-test-warnings-0.031-pl5321ha770c72_0.conda + sha256: 2aaea17941fd456bfe44d15e3f3651d7f27be0cb31235d88e48b71ae857d7ef1 + md5: db0eb51272ea7af7213afaaf7e9967c3 + depends: + - perl >=5.32.1,<5.33.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 21724 + timestamp: 1669240328383 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-test2-suite-0.000163-pl5321hdfd78af_0.tar.bz2 + sha256: 226cbad887607747b19816100d00b0e3dcca66b1e2b282efc1aad225eaacd27d + md5: 6a29ce25ffd66ecc495a62a77a4dc0c2 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-importer + - perl-scope-guard + - perl-sub-info + - perl-term-table + license: perl_5 + size: 212676 + timestamp: 1717604907839 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-try-tiny-0.31-pl5321ha770c72_0.tar.bz2 + sha256: 7e74c95eb085b095a258a22ee2f30192d79f23a58cbbcb8ca2b14f4fe0a8c406 + md5: cc23b14ed56bf51d84832d98ebd156af + depends: + - perl >=5.32.1,<5.33.0a0 *_perl5 + license: MIT + license_family: MIT + size: 17704 + timestamp: 1659695570098 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-types-serialiser-1.01-pl5321hdfd78af_0.tar.bz2 + sha256: 20f61217b16235d0161ad6fa0a234585afcc04ec5a6142c65b5867e26216dfe3 + md5: cfc65753e827bbef80c00eaa395f6ae7 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-common-sense + license: perl_5 + size: 13136 + timestamp: 1644512391683 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/perl-uri-5.34-pl5321ha770c72_0.conda + sha256: b8ee5b5b4a654ce41e9c676255f8af3e52b448ddc3473b72fe3eba387a9f3b46 + md5: dc588517ea63f0d7a352c2b5783c1b58 + depends: + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-business-isbn + - perl-test-fatal 0.016.* + - perl-test-warnings 0.031.* + license: GPL-1.0-or-later OR Artistic-1.0-Perl + size: 78694 + timestamp: 1758130476076 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/perl-xml-libxml-2.0210-pl5321hf886d80_0.tar.bz2 + sha256: 0a93d64da53635e81a06c363081b80d29598b4fd1e45b30410b2e6898c4140e4 + md5: ee29e9d7e2f57fa48b6da8ea82ab9e61 + depends: + - libgcc >=13 + - libxml2 >=2.13.5,<3.0a0 + - libzlib >=1.3.1,<2.0a0 + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-alien-build >=2.84,<3.0a0 + - perl-alien-libxml2 >=0.17,<0.18.0a0 + - perl-xml-namespacesupport + - perl-xml-sax + - zlib + license: Perl + size: 277975 + timestamp: 1735914218064 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-xml-namespacesupport-1.12-pl5321hd8ed1ab_0.tar.bz2 + sha256: bf384bea1057085fdf8daac6e67a96ca892f2fedf4ce0327f9bda5d3a9bce102 + md5: f4ce684ba66b3228bfffb09892235930 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-constant + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 22890 + timestamp: 1664723687849 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-xml-sax-1.02-pl5321hd8ed1ab_0.tar.bz2 + sha256: 324ca36bdfff272c8ac06befcd9d1e9e1b22c1c6449bb3801d7e8b9c92acc6b6 + md5: 3e3fac6ffef3fcda271b5511aecacaa8 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-file-temp + - perl-xml-namespacesupport + - perl-xml-sax-base + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 43526 + timestamp: 1664728646828 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/perl-xml-sax-base-1.09-pl5321hd8ed1ab_0.tar.bz2 + sha256: e652410a4d02bde1a1d7e0021ee84006113785676267c1e33642ba3d2ea9c254 + md5: a4d9b02cd61c478c9cb58064307a1414 + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: GPL-1.0-or-later OR Artistic-1.0-Perl + license_family: OTHER + size: 27575 + timestamp: 1664556675309 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/perl-yaml-1.30-pl5321hdfd78af_0.tar.bz2 + noarch: true + sha256: 6813626ecc8859b4f16de20f35c9bfdfbdf0a60d7da33494bca6c0c34511a76e + md5: da9ef80f4c4c2374d7f4ec7af6d96b6a + depends: + - perl >=5.32.1,<6.0a0 *_perl5 + license: perl_5 + size: 43572 + timestamp: 1644444148032 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pillow-12.0.0-py311h07c5bb8_0.conda + sha256: 57231a713744270bcd7116f339e13c78cd78f055a54b4d9b811a8597076c21d2 + md5: 51f505a537b2d216a1b36b823df80995 + depends: + - python + - libgcc >=14 + - __glibc >=2.17,<3.0.a0 + - tk >=8.6.13,<8.7.0a0 + - libxcb >=1.17.0,<2.0a0 + - libjpeg-turbo >=3.1.0,<4.0a0 + - python_abi 3.11.* *_cp311 + - openjpeg >=2.5.4,<3.0a0 + - libfreetype >=2.14.1 + - libfreetype6 >=2.14.1 + - zlib-ng >=2.2.5,<2.3.0a0 + - libwebp-base >=1.6.0,<2.0a0 + - lcms2 >=2.17,<3.0a0 + - libtiff >=4.7.1,<4.8.0a0 + license: HPND + size: 1044368 + timestamp: 1761655794832 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pip-25.3-pyh8b19718_0.conda + sha256: b67692da1c0084516ac1c9ada4d55eaf3c5891b54980f30f3f444541c2706f1e + md5: c55515ca43c6444d2572e0f0d93cb6b9 + depends: + - python >=3.10,<3.13.0a0 + - setuptools + - wheel + license: MIT + license_family: MIT + size: 1177534 + timestamp: 1762776258783 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pixman-0.46.4-h54a6638_1.conda + sha256: 43d37bc9ca3b257c5dd7bf76a8426addbdec381f6786ff441dc90b1a49143b6a + md5: c01af13bdc553d1a8fbfff6e8db075f0 + depends: + - libgcc >=14 + - libstdcxx >=14 + - libgcc >=14 + - __glibc >=2.17,<3.0.a0 + license: MIT + license_family: MIT + size: 450960 + timestamp: 1754665235234 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/platformdirs-4.5.0-pyhcf101f3_0.conda + sha256: 7efd51b48d908de2d75cbb3c4a2e80dd9454e1c5bb8191b261af3136f7fa5888 + md5: 5c7a868f8241e64e1cf5fdf4962f23e2 + depends: + - python >=3.10 + - python + license: MIT + license_family: MIT + size: 23625 + timestamp: 1759953252315 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pluggy-1.6.0-pyhd8ed1ab_0.conda + sha256: a8eb555eef5063bbb7ba06a379fa7ea714f57d9741fe0efdb9442dbbc2cccbcc + md5: 7da7ccd349dbf6487a7778579d2bb971 + depends: + - python >=3.9 + license: MIT + license_family: MIT + size: 24246 + timestamp: 1747339794916 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/ply-3.11-pyhd8ed1ab_3.conda + sha256: bae453e5cecf19cab23c2e8929c6e30f4866d996a8058be16c797ed4b935461f + md5: fd5062942bfa1b0bd5e0d2a4397b099e + depends: + - python >=3.9 + license: BSD-3-Clause + license_family: BSD + size: 49052 + timestamp: 1733239818090 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/pplacer-1.1.alpha19-h9ee0642_2.tar.bz2 + sha256: e4f66428e1c5dd1580031360dc0fcd702a08edc48e3c95e5520e34c61acb141a + md5: 9c132838ff736fc8c11edfe81699a4c1 + license: GPL-3.0 + license_family: GPL + size: 9022058 + timestamp: 1616609356153 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/prank-170427-h9948957_1.tar.bz2 + sha256: 21a20fa1b3ebc52e527bcef7261fbdab3cd29c76ad239446ea18bfbc652ffb6c + md5: b42e651a06117c624a977b6a24b1006d + depends: + - libgcc >=13 + - libstdcxx >=13 + license: GPL-3.0-or-later + license_family: GPL3 + size: 410215 + timestamp: 1733803138547 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/prodigal-2.6.3-h577a1d6_11.tar.bz2 + sha256: 1211d9f01128f141154cb4616aa5b15890afc26699767886f6655cabfd67ec06 + md5: 7b083f573760cbd88a206e05f73f5e9d + depends: + - libgcc >=13 + - libzlib >=1.3.1,<2.0a0 + license: GPL-3.0-or-later + license_family: GPL3 + size: 601984 + timestamp: 1752712731224 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pthread-stubs-0.4-hb9d3cd8_1002.conda + sha256: 9c88f8c64590e9567c6c80823f0328e58d3b1efb0e1c539c0315ceca764e0973 + md5: b3c17d95b5a10c6e64a21fa17573e70e + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: MIT + license_family: MIT + size: 8252 + timestamp: 1726802366959 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pulseaudio-client-17.0-h9a8bead_2.conda + sha256: 8a6729861c9813a756b0438c30bd271722fb3f239ded3afc3bf1cb03327a640e + md5: b6f21b1c925ee2f3f7fc37798c5988db + depends: + - __glibc >=2.17,<3.0.a0 + - dbus >=1.16.2,<2.0a0 + - libgcc >=14 + - libglib >=2.86.0,<3.0a0 + - libiconv >=1.18,<2.0a0 + - libsndfile >=1.2.2,<1.3.0a0 + - libsystemd0 >=257.7 + - libxcb >=1.17.0,<2.0a0 + constrains: + - pulseaudio 17.0 *_2 + license: LGPL-2.1-or-later + license_family: LGPL + size: 761857 + timestamp: 1757472971364 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pybind11-abi-4-hd8ed1ab_3.tar.bz2 + sha256: d4fb485b79b11042a16dc6abfb0c44c4f557707c2653ac47c81e5d32b24a3bb0 + md5: 878f923dd6acc8aeb47a75da6c4098be + license: BSD-3-Clause + license_family: BSD + size: 9906 + timestamp: 1610372835205 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pycosat-0.6.6-py311h49ec1c0_3.conda + sha256: 61c07e45a0a0c7a2b0dc986a65067fc2b00aba51663b7b05d4449c7862d7a390 + md5: 77c1b47af5775a813193f7870be8644a + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: MIT + license_family: MIT + size: 88491 + timestamp: 1757744790912 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pycparser-2.22-pyh29332c3_1.conda + sha256: 79db7928d13fab2d892592223d7570f5061c192f27b9febd1a418427b719acc6 + md5: 12c566707c80111f9799308d9e265aef + depends: + - python >=3.9 + - python + license: BSD-3-Clause + license_family: BSD + size: 110100 + timestamp: 1733195786147 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pygments-2.19.2-pyhd8ed1ab_0.conda + sha256: 5577623b9f6685ece2697c6eb7511b4c9ac5fb607c9babc2646c811b428fd46a + md5: 6b6ece66ebcae2d5f326c77ef2c5a066 + depends: + - python >=3.9 + license: BSD-2-Clause + license_family: BSD + size: 889287 + timestamp: 1750615908735 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pygtrie-2.5.0-pyhd8ed1ab_1.conda + sha256: c4f840c064e62d697a6e55fb05585ad9f22e1b21b53a45bd6c27f60e134bafb8 + md5: ffa9a72b339edc8856c3fc17e8d74128 + depends: + - python >=3.9 + license: Apache 2.0 + license_family: Apache + size: 31912 + timestamp: 1734664644850 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pymongo-4.15.4-py311h1ddb823_0.conda + sha256: ec54d6a0ae50b9271fb28b47b6f18f743e66f893b9d53d7ed327890e820d15d4 + md5: 36f445f69e27e7ed90d70722f87fda8f + depends: + - __glibc >=2.17,<3.0.a0 + - dnspython <3.0.0,>=1.16.0 + - libgcc >=14 + - libstdcxx >=14 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: Apache-2.0 + size: 2327339 + timestamp: 1763048682599 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pyparsing-3.2.5-pyhcf101f3_0.conda + sha256: 6814b61b94e95ffc45ec539a6424d8447895fef75b0fec7e1be31f5beee883fb + md5: 6c8979be6d7a17692793114fa26916e8 + depends: + - python >=3.10 + - python + license: MIT + license_family: MIT + size: 104044 + timestamp: 1758436411254 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pyqt-5.15.11-py311h0580839_2.conda + sha256: 5066cbba17b271b62e8c290994a312217a47c5e23259be1ef700ffaed2646221 + md5: 59ae5d8d4bcb1371d61ec49dfb985c70 + depends: + - __glibc >=2.17,<3.0.a0 + - libegl >=1.7.0,<2.0a0 + - libgcc >=14 + - libgl >=1.7.0,<2.0a0 + - libopengl >=1.7.0,<2.0a0 + - libstdcxx >=14 + - pyqt5-sip 12.17.0 py311h1ddb823_2 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + - qt-main >=5.15.15,<5.16.0a0 + - sip >=6.10.0,<6.11.0a0 + - xcb-util >=0.4.1,<0.5.0a0 + - xcb-util-image >=0.4.0,<0.5.0a0 + - xcb-util-keysyms >=0.4.1,<0.5.0a0 + - xcb-util-renderutil >=0.3.10,<0.4.0a0 + - xcb-util-wm >=0.4.2,<0.5.0a0 + - xorg-libice >=1.1.2,<2.0a0 + - xorg-libsm >=1.2.6,<2.0a0 + - xorg-libx11 >=1.8.12,<2.0a0 + - xorg-libxcomposite >=0.4.6,<1.0a0 + - xorg-libxdamage >=1.1.6,<2.0a0 + - xorg-libxext >=1.3.6,<2.0a0 + - xorg-libxxf86vm >=1.1.6,<2.0a0 + license: GPL-3.0-only + license_family: GPL + size: 5217528 + timestamp: 1759497952060 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pyqt5-sip-12.17.0-py311h1ddb823_2.conda + sha256: 106d5894a0ff3ba892c10a1ffed5bf05583c2a4b29f8e62fe90eed71274dfb05 + md5: 4f296d802e51e7a6889955c7f1bd10be + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - packaging + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + - sip + - toml + license: GPL-3.0-only + license_family: GPL + size: 85010 + timestamp: 1759495564200 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pyside6-6.9.2-py311h72d58bf_1.conda + sha256: 9540cbc6de195e8fb49ea4ef39567f11982580f007914617e1ab482193db114c + md5: 4d8a5ee88cbf101a97b129eec7042af9 + depends: + - __glibc >=2.17,<3.0.a0 + - libclang13 >=21.1.0 + - libegl >=1.7.0,<2.0a0 + - libgcc >=14 + - libgl >=1.7.0,<2.0a0 + - libopengl >=1.7.0,<2.0a0 + - libstdcxx >=14 + - libxml2 >=2.13.8,<2.14.0a0 + - libxslt >=1.1.43,<2.0a0 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + - qt6-main 6.9.2.* + - qt6-main >=6.9.2,<6.10.0a0 + license: LGPL-3.0-only + license_family: LGPL + size: 10150360 + timestamp: 1756675160062 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pysocks-1.7.1-pyha55dd90_7.conda + sha256: ba3b032fa52709ce0d9fd388f63d330a026754587a2f461117cac9ab73d8d0d8 + md5: 461219d1a5bd61342293efa2c0c90eac + depends: + - __unix + - python >=3.9 + license: BSD-3-Clause + license_family: BSD + size: 21085 + timestamp: 1733217331982 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/python-3.11.14-hd63d673_2_cpython.conda + build_number: 2 + sha256: 5b872f7747891e50e990a96d2b235236a5c66cc9f8c9dcb7149aee674ea8145a + md5: c4202a55b4486314fbb8c11bc43a29a0 + depends: + - __glibc >=2.17,<3.0.a0 + - bzip2 >=1.0.8,<2.0a0 + - ld_impl_linux-64 >=2.36.1 + - libexpat >=2.7.1,<3.0a0 + - libffi >=3.5.2,<3.6.0a0 + - libgcc >=14 + - liblzma >=5.8.1,<6.0a0 + - libnsl >=2.0.1,<2.1.0a0 + - libsqlite >=3.50.4,<4.0a0 + - libuuid >=2.41.2,<3.0a0 + - libxcrypt >=4.4.36 + - libzlib >=1.3.1,<2.0a0 + - ncurses >=6.5,<7.0a0 + - openssl >=3.5.4,<4.0a0 + - readline >=8.2,<9.0a0 + - tk >=8.6.13,<8.7.0a0 + - tzdata + constrains: + - python_abi 3.11.* *_cp311 + license: Python-2.0 + size: 30874708 + timestamp: 1761174520369 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda + sha256: d6a17ece93bbd5139e02d2bd7dbfa80bee1a4261dced63f65f679121686bf664 + md5: 5b8d21249ff20967101ffa321cab24e8 + depends: + - python >=3.9 + - six >=1.5 + - python + license: Apache-2.0 + license_family: APACHE + size: 233310 + timestamp: 1751104122689 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/python-tzdata-2025.2-pyhd8ed1ab_0.conda + sha256: e8392a8044d56ad017c08fec2b0eb10ae3d1235ac967d0aab8bd7b41c4a5eaf0 + md5: 88476ae6ebd24f39261e0854ac244f33 + depends: + - python >=3.9 + license: Apache-2.0 + license_family: APACHE + size: 144160 + timestamp: 1742745254292 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/python_abi-3.11-8_cp311.conda + build_number: 8 + sha256: fddf123692aa4b1fc48f0471e346400d9852d96eeed77dbfdd746fa50a8ff894 + md5: 8fcb6b0e2161850556231336dae58358 + constrains: + - python 3.11.* *_cpython + license: BSD-3-Clause + license_family: BSD + size: 7003 + timestamp: 1752805919375 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/pytz-2025.2-pyhd8ed1ab_0.conda + sha256: 8d2a8bf110cc1fc3df6904091dead158ba3e614d8402a83e51ed3a8aa93cdeb0 + md5: bc8e3267d44011051f2eb14d22fb0960 + depends: + - python >=3.9 + license: MIT + license_family: MIT + size: 189015 + timestamp: 1742920947249 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/pyyaml-6.0.3-py311h3778330_0.conda + sha256: 7dc5c27c0c23474a879ef5898ed80095d26de7f89f4720855603c324cca19355 + md5: 707c3d23f2476d3bfde8345b4e7d7853 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + - yaml >=0.2.5,<0.3.0a0 + license: MIT + license_family: MIT + size: 211606 + timestamp: 1758892088237 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/qhull-2020.2-h434a139_5.conda + sha256: 776363493bad83308ba30bcb88c2552632581b143e8ee25b1982c8c743e73abc + md5: 353823361b1d27eb3960efb076dfcaf6 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc-ng >=12 + - libstdcxx-ng >=12 + license: LicenseRef-Qhull + size: 552937 + timestamp: 1720813982144 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/qt-main-5.15.15-h3a7ef08_5.conda + sha256: f1fee8d35bfeb4806bdf2cb13dc06e91f19cb40104e628dd721989885d1747ad + md5: 9279a2436ad1ba296f49f0ad44826b78 + depends: + - __glibc >=2.17,<3.0.a0 + - alsa-lib >=1.2.14,<1.3.0a0 + - dbus >=1.16.2,<2.0a0 + - fontconfig >=2.15.0,<3.0a0 + - fonts-conda-ecosystem + - gst-plugins-base >=1.24.11,<1.25.0a0 + - gstreamer >=1.24.11,<1.25.0a0 + - harfbuzz >=11.4.3 + - icu >=75.1,<76.0a0 + - krb5 >=1.21.3,<1.22.0a0 + - libclang-cpp20.1 >=20.1.8,<20.2.0a0 + - libclang13 >=20.1.8 + - libcups >=2.3.3,<2.4.0a0 + - libdrm >=2.4.125,<2.5.0a0 + - libegl >=1.7.0,<2.0a0 + - libevent >=2.1.12,<2.1.13.0a0 + - libexpat >=2.7.1,<3.0a0 + - libfreetype >=2.13.3 + - libfreetype6 >=2.13.3 + - libgcc >=13 + - libgl >=1.7.0,<2.0a0 + - libglib >=2.84.3,<3.0a0 + - libjpeg-turbo >=3.1.0,<4.0a0 + - libllvm20 >=20.1.8,<20.2.0a0 + - libpng >=1.6.50,<1.7.0a0 + - libpq >=17.6,<18.0a0 + - libsqlite >=3.50.4,<4.0a0 + - libstdcxx >=13 + - libxcb >=1.17.0,<2.0a0 + - libxkbcommon >=1.11.0,<2.0a0 + - libxml2 >=2.13.8,<2.14.0a0 + - libzlib >=1.3.1,<2.0a0 + - nspr >=4.37,<5.0a0 + - nss >=3.115,<4.0a0 + - openssl >=3.5.2,<4.0a0 + - pulseaudio-client >=17.0,<17.1.0a0 + - xcb-util >=0.4.1,<0.5.0a0 + - xcb-util-image >=0.4.0,<0.5.0a0 + - xcb-util-keysyms >=0.4.1,<0.5.0a0 + - xcb-util-renderutil >=0.3.10,<0.4.0a0 + - xcb-util-wm >=0.4.2,<0.5.0a0 + - xorg-libice >=1.1.2,<2.0a0 + - xorg-libsm >=1.2.6,<2.0a0 + - xorg-libx11 >=1.8.12,<2.0a0 + - xorg-libxdamage >=1.1.6,<2.0a0 + - xorg-libxext >=1.3.6,<2.0a0 + - xorg-libxxf86vm >=1.1.6,<2.0a0 + - zstd >=1.5.7,<1.6.0a0 + constrains: + - qt 5.15.15 + license: LGPL-3.0-only + license_family: LGPL + size: 52149940 + timestamp: 1756072007197 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/qt6-main-6.9.2-h5bd77bc_1.conda + sha256: ac540c33b8e908f49e4eae93032708f7f6eeb5016d28190f6ed7543532208be2 + md5: f7bfe5b8e7641ce7d11ea10cfd9f33cc + depends: + - __glibc >=2.17,<3.0.a0 + - alsa-lib >=1.2.14,<1.3.0a0 + - dbus >=1.16.2,<2.0a0 + - double-conversion >=3.3.1,<3.4.0a0 + - fontconfig >=2.15.0,<3.0a0 + - fonts-conda-ecosystem + - harfbuzz >=11.5.0 + - icu >=75.1,<76.0a0 + - krb5 >=1.21.3,<1.22.0a0 + - libclang-cpp21.1 >=21.1.0,<21.2.0a0 + - libclang13 >=21.1.0 + - libcups >=2.3.3,<2.4.0a0 + - libdrm >=2.4.125,<2.5.0a0 + - libegl >=1.7.0,<2.0a0 + - libfreetype >=2.14.1 + - libfreetype6 >=2.14.1 + - libgcc >=14 + - libgl >=1.7.0,<2.0a0 + - libglib >=2.86.0,<3.0a0 + - libjpeg-turbo >=3.1.0,<4.0a0 + - libllvm21 >=21.1.0,<21.2.0a0 + - libpng >=1.6.50,<1.7.0a0 + - libpq >=17.6,<18.0a0 + - libsqlite >=3.50.4,<4.0a0 + - libstdcxx >=14 + - libtiff >=4.7.0,<4.8.0a0 + - libwebp-base >=1.6.0,<2.0a0 + - libxcb >=1.17.0,<2.0a0 + - libxkbcommon >=1.11.0,<2.0a0 + - libxml2 >=2.13.8,<2.14.0a0 + - libzlib >=1.3.1,<2.0a0 + - openssl >=3.5.2,<4.0a0 + - pcre2 >=10.46,<10.47.0a0 + - wayland >=1.24.0,<2.0a0 + - xcb-util >=0.4.1,<0.5.0a0 + - xcb-util-cursor >=0.1.5,<0.2.0a0 + - xcb-util-image >=0.4.0,<0.5.0a0 + - xcb-util-keysyms >=0.4.1,<0.5.0a0 + - xcb-util-renderutil >=0.3.10,<0.4.0a0 + - xcb-util-wm >=0.4.2,<0.5.0a0 + - xorg-libice >=1.1.2,<2.0a0 + - xorg-libsm >=1.2.6,<2.0a0 + - xorg-libx11 >=1.8.12,<2.0a0 + - xorg-libxcomposite >=0.4.6,<1.0a0 + - xorg-libxcursor >=1.2.3,<2.0a0 + - xorg-libxdamage >=1.1.6,<2.0a0 + - xorg-libxext >=1.3.6,<2.0a0 + - xorg-libxrandr >=1.5.4,<2.0a0 + - xorg-libxtst >=1.2.5,<2.0a0 + - xorg-libxxf86vm >=1.1.6,<2.0a0 + - zstd >=1.5.7,<1.6.0a0 + constrains: + - qt 6.9.2 + license: LGPL-3.0-only + license_family: LGPL + size: 52405921 + timestamp: 1758011263853 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-abind-1.4_8-r44hc72bb7e_1.conda + sha256: 1d83d808a9a52b1cb3919b110687d2fcb605e96143136745cca9bb7c44e4b06f + md5: bc7c760b0b0ad2a6eb8970f56ef82971 + depends: + - r-base >=4.4,<4.5.0a0 + license: LGPL (>= 2) + license_family: LGPL + size: 82485 + timestamp: 1757460279230 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-amap-0.8_20-r44ha36cffa_1.conda + sha256: b40882a64798b805490fb73b4fd4ac65274185a999dfee9ca5c5af30af98f5cc + md5: 30c93d45ec4b429f1116afe40e8ee808 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libgfortran + - libgfortran5 >=14.3.0 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL3 + size: 303693 + timestamp: 1758171087251 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-ape-5.8_1-r44h3704496_2.conda + sha256: 590851bd97ff95698140031da1e9046f17f04448ebc24486382d7f68b3177827 + md5: 2dc94a46f0238174b5de6edc693a3c92 + depends: + - __glibc >=2.17,<3.0.a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=14 + - liblapack >=3.9.0,<4.0a0 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-digest + - r-lattice + - r-nlme + - r-rcpp >=0.12.0 + license: GPL-2.0-or-later + license_family: GPL3 + size: 2941194 + timestamp: 1757480722386 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-argparse-2.3.1-r44hc72bb7e_0.conda + sha256: 8ed645cc4d2ed29c135e0672428759b2c35054d651d039f4d11bf78e59a79130 + md5: 53b38ec057737379504745d4108a3ac3 + depends: + - python >=3.2 + - r-base >=4.4,<4.5.0a0 + - r-findpython + - r-jsonlite + - r-r6 + license: GPL-2.0-or-later + license_family: GPL3 + size: 188072 + timestamp: 1759962316617 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-askpass-1.2.1-r44h54b55ab_1.conda + sha256: 39a8bdb086df98cfeeedadeeafdbfcd8a5b90a3c266229d1ab7a2b25fc1db2b3 + md5: ae87c9a5af5a2ebfa5412037825f4dd2 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + - r-sys >=2.1 + license: MIT + license_family: MIT + size: 32088 + timestamp: 1758383484942 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-assertthat-0.2.1-r44hc72bb7e_6.conda + sha256: 7ce5cad4870512fce1cec74f88673ea84ba0e4b00b872f4c0b1a937b44690e71 + md5: a9e34b8723a9a38a0fb18746ad30a546 + depends: + - r-base >=4.4,<4.5.0a0 + license: GPL-3.0-only + license_family: GPL3 + size: 72978 + timestamp: 1757447415952 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-backports-1.5.0-r44h54b55ab_2.conda + sha256: 3cfda6bcdac741cc9ced0c21fbe2d9b19ebb43cbf5759d158fd82ff802a22b5e + md5: be42ed7e546105353f6acb427d739f02 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL2 + size: 131556 + timestamp: 1757441849524 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-base-4.4.3-hc038350_5.conda + sha256: a102049390e9cfe88996bc028717aac40131397d2748ed2e0ce4efa94a5664df + md5: 6a0beb99d5f3767b373bf21b4a90939a + depends: + - __glibc >=2.17,<3.0.a0 + - _openmp_mutex >=4.5 + - _r-mutex 1.* anacondar_1 + - bwidget + - bzip2 >=1.0.8,<2.0a0 + - cairo >=1.18.4,<2.0a0 + - curl + - gcc_impl_linux-64 >=10 + - gfortran_impl_linux-64 + - gsl >=2.7,<2.8.0a0 + - gxx_impl_linux-64 >=10 + - icu >=75.1,<76.0a0 + - libblas >=3.9.0,<4.0a0 + - libcurl >=8.17.0,<9.0a0 + - libdeflate >=1.25,<1.26.0a0 + - libexpat >=2.7.1,<3.0a0 + - libgcc + - libgcc-ng >=12 + - libgfortran + - libgfortran-ng + - libgfortran5 >=10.4.0 + - libglib >=2.86.1,<3.0a0 + - libiconv >=1.18,<2.0a0 + - libjpeg-turbo >=3.1.2,<4.0a0 + - liblapack >=3.9.0,<4.0a0 + - liblzma >=5.8.1,<6.0a0 + - libpng >=1.6.50,<1.7.0a0 + - libstdcxx + - libstdcxx-ng >=12 + - libtiff >=4.7.1,<4.8.0a0 + - libuuid >=2.41.2,<3.0a0 + - libzlib >=1.3.1,<2.0a0 + - make + - pango >=1.56.4,<2.0a0 + - pcre2 >=10.46,<10.47.0a0 + - readline >=8.2,<9.0a0 + - sed + - tk >=8.6.13,<8.7.0a0 + - tktable + - tzdata >=2024a + - xorg-libxt + license: GPL-2.0-or-later + license_family: GPL + size: 27137250 + timestamp: 1762369142083 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-base64enc-0.1_3-r44h54b55ab_1008.conda + sha256: da98331e2c2149252ff217b9f919c3ddf464fa6321c9399ce6530891af53424f + md5: cf6a4cdd40b10ef6d15f1c9ee18010a9 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL3 + size: 46444 + timestamp: 1757421757586 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-bh-1.87.0_1-r44hc72bb7e_1.conda + sha256: 564bb989ec8dbf572876e9cd1c84de670c5168a7bb107d2043fc33cf5a80c183 + md5: b7c43e2c7fd3f9b9599123ea9a4be21d + depends: + - r-base >=4.4,<4.5.0a0 + license: BSL-1.0 + license_family: OTHER + size: 11611316 + timestamp: 1757467893201 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-biasedurn-2.0.12-r44h3697838_2.conda + sha256: 96eb3baf66f8a7b05148e588ef64e7956818ff4d6bb54848e41bcb45c0469bdd + md5: 9244a0ba2744bc31fed8cbf7d076b5a2 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-3.0-only + license_family: GPL3 + size: 304115 + timestamp: 1757817771760 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-bit-4.6.0-r44h54b55ab_1.conda + sha256: 66f050183d8ffd9de6c4b763f43109490196f2a11dfad98e9e8892175aaa6263 + md5: 4e905067722f1416de735a3d6b33825c + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL2 + size: 621865 + timestamp: 1757441702226 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-bit64-4.6.0_1-r44h54b55ab_1.conda + sha256: dfb04a9bee5bc1efd51ec567cf47d90fb997c8040bebf9d4c7c01cec6ff58507 + md5: 6c45d3611b0fe56dfc2276a7a9a2a87d + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + - r-bit >=4.0.0 + license: GPL-2.0-only + license_family: GPL2 + size: 504762 + timestamp: 1757457026892 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-bitops-1.0_9-r44h54b55ab_1.conda + sha256: 39ff41868e21cbf57e23838ae6b7f68d9bc339f9aeae027c19d1cbdcfe70c06f + md5: 7033ada947ced7f3b2a155f20a0f8645 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL3 + size: 45778 + timestamp: 1757447616166 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-blob-1.2.4-r44hc72bb7e_3.conda + sha256: 0364f8e079240797026967830e315b4f3b9aa03967cbc56805f942e1acfeceff + md5: 5af514da92a8df270f9890f36961f492 + depends: + - r-base >=4.4,<4.5.0a0 + - r-rlang + - r-vctrs >=0.2.1 + license: GPL-3.0-only + license_family: GPL3 + size: 67835 + timestamp: 1757493686018 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-bms-0.3.5-r44hc72bb7e_4.conda + sha256: ad183f1ce11027af56e43cb9dfe2b8fede89f5a60b7536e30151f5c5c89d269a + md5: 12a311548e8fc247279e07ef231f8263 + depends: + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + license_family: OTHER + size: 2928553 + timestamp: 1757923022257 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-brew-1.0_10-r44hc72bb7e_2.conda + sha256: e77bc79cd89cf3c106e69954f82f3483b6ec89c547197a09b7795449ba86b56a + md5: 03ab88ce441b876e359ab575b40ff11b + depends: + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-only + license_family: GPL2 + size: 69079 + timestamp: 1757447878952 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-broom-1.0.10-r44hc72bb7e_0.conda + sha256: 7a3bdbad128abda37e267aed6ab6670b0e0dec918650c53dac18087a72f22e10 + md5: e80c1c49996f186133d6725c9257d9c4 + depends: + - r-backports + - r-base >=4.4,<4.5.0a0 + - r-dplyr >=1.0.0 + - r-ellipsis + - r-generics >=0.0.2 + - r-ggplot2 + - r-glue + - r-purrr + - r-rlang + - r-stringr + - r-tibble >=3.0.0 + - r-tidyr >=1.0.0 + license: MIT + license_family: MIT + size: 1877258 + timestamp: 1757769946164 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-bslib-0.9.0-r44hc72bb7e_1.conda + sha256: 0bf3d6885c018227faf057d52c0760a58ce040d907c7eafef5c4c61149b83d15 + md5: a0e55df03531b9dec1a2a50850855802 + depends: + - r-base >=4.4,<4.5.0a0 + - r-base64enc + - r-cachem + - r-htmltools >=0.5.7 + - r-jquerylib >=0.1.3 + - r-jsonlite + - r-lifecycle + - r-memoise >=2.0.1 + - r-mime + - r-rlang + - r-sass >=0.4.0 + license: MIT + license_family: MIT + size: 5041215 + timestamp: 1757484861965 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-cachem-1.1.0-r44h54b55ab_2.conda + sha256: 1a68b04169a635a3890573cd3f07a791ca1b27b1a1171109b0402a2f8787162d + md5: 9e798e9474af81b679133429db82de66 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + - r-fastmap + - r-rlang + license: MIT + license_family: MIT + size: 76618 + timestamp: 1757441491774 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-callr-3.7.6-r44hc72bb7e_2.conda + sha256: e85f816c249baa83e8bbfd8aa15ef9249151d9aaae69c5010ea209e8d1fa3ff9 + md5: 020dde50545870904c62062ffab4b621 + depends: + - r-base >=4.4,<4.5.0a0 + - r-processx >=3.4.0 + - r-r6 + license: MIT + license_family: MIT + size: 454467 + timestamp: 1757475630913 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-catools-1.18.3-r44h3697838_1.conda + sha256: e5d962a05b846a404df304364acd72316771934e77459e01ee38f5b3ed19f29a + md5: 78c11c13b11bba537d9414d3d9d741aa + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-bitops + license: GPL-3.0-only + license_family: GPL3 + size: 226174 + timestamp: 1757495084182 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-cellranger-1.1.0-r44hc72bb7e_1008.conda + sha256: 042dda928eb65997abaa2233f12b7a8c780adb00755f6b9840cca49250263a3b + md5: 2dcdfa8fa82d4dfef20cdbcb838ec89f + depends: + - r-base >=4.4,<4.5.0a0 + - r-rematch + - r-tibble + license: MIT + license_family: MIT + size: 111786 + timestamp: 1757511361046 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-checkmate-2.3.3-r44h54b55ab_1.conda + sha256: 26458ac4dbb31bc708c00bf5f08fe1e7b2c594cfd5d96c5db4beee6a9eab6747 + md5: 257a9a3c414ce1ff6cf908b0b5f44098 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-backports >=1.1.0 + - r-base >=4.4,<4.5.0a0 + license: BSD-3-Clause + license_family: BSD + size: 730755 + timestamp: 1757465716605 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-cli-3.6.5-r44h3697838_1.conda + sha256: 93ab8089d7e406c67b5bb0d47c6d45196be0d1e84a4aa0198fad48c19cad56c2 + md5: ae71dbdd32ef9384c43cfb4e3991305a + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 1314296 + timestamp: 1757414989705 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-clipr-0.8.0-r44hc72bb7e_4.conda + sha256: dc7693e0fa3e16290bc5eca5d918e4fa057aedab64520cf4c9055096b988c2be + md5: 1f404af69237bf7d0dda28b53c723b9c + depends: + - r-base >=4.4,<4.5.0a0 + license: GPL-3.0-only + license_family: GPL3 + size: 70741 + timestamp: 1757460201245 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-cluster-2.1.8.1-r44heaba542_1.conda + sha256: 9bbce167b7d0e6968f3eacc3e0c606e073a47d97936c8755801077723b420844 + md5: e15e4e8023eeae7e9ff3f6666ae04eb5 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libgfortran + - libgfortran5 >=14.3.0 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL3 + size: 591946 + timestamp: 1757458045161 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-codetools-0.2_20-r44hc72bb7e_2.conda + sha256: 3cf1c320913ddae7bd91ba770535a9d4765862a9fb1bac419a4ea41b8c39d693 + md5: d81c629f78e571b14f9a617253c44552 + depends: + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL + size: 109324 + timestamp: 1757452124857 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-collections-0.3.9-r44h54b55ab_1.conda + sha256: a7a51ddee74de436f62628ce1c589b068214c5ae269e323eef93d6656ea17a45 + md5: 7d5e5f8e03d0a2b061be0413ca7e6084 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 76503 + timestamp: 1757422267175 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-colorspace-2.1_2-r44h54b55ab_0.conda + sha256: 31ce342cdaccc3ec17d988e13d06d61ce7f6684552b2b86c513f7b547978ce1f + md5: 209970b5b939d6f778c78aa304d3df4a + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: BSD-3-Clause + license_family: BSD + size: 2543924 + timestamp: 1758590585832 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-commonmark-2.0.0-r44h54b55ab_1.conda + sha256: 3476f88b4aad66847acaf674739abb8d382129e85272e24667887424c0d47934 + md5: 9aed52e98aeac71945621364047ab682 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: BSD-2-Clause + license_family: BSD + size: 139988 + timestamp: 1757422083741 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-conflicted-1.2.0-r44h785f33e_3.conda + sha256: 8f7577f5a75e9893e5b80dc2b79274beae628cce7e9bc0088c8fee502e143d14 + md5: ec2f4608ada6971bbdfa67867951d713 + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli >=3.4.0 + - r-memoise + - r-rlang >=1.0.0 + license: MIT + license_family: MIT + size: 63899 + timestamp: 1757548489611 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-cpp11-0.5.2-r44h785f33e_2.conda + sha256: d2a017e0127b7b84745e2ceb050af4f36383b90484a27e962767c854eb83e9b4 + md5: 31026a9cca367464a845387c0b5f0d55 + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 236536 + timestamp: 1757451995134 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-crayon-1.5.3-r44hc72bb7e_2.conda + sha256: 93bb26deef54d51293c3582cfbb51987a7af6ca93d498c4ffbe8ebdcfe52cad0 + md5: 2b72da5cca65460ba6ac1f81b4ff72cc + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 167903 + timestamp: 1757452343297 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-curl-7.0.0-r44h10955f1_1.conda + sha256: aa0a725d12e11d372074217e693404917a2be1a576e1a4538b4b54a149874964 + md5: 69e006c582c9653920be9b42c9a5d325 + depends: + - __glibc >=2.17,<3.0.a0 + - libcurl >=8.14.1,<9.0a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 480191 + timestamp: 1757581715910 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-cyclocomp-1.1.1-r44hc72bb7e_2.conda + sha256: f6308cb29686cec31542236d0234dcc68548af7d94cda9360e4b8c602b59569e + md5: 78ac490dafc5a9dd078702802a511665 + depends: + - r-base >=4.4,<4.5.0a0 + - r-callr + - r-crayon + - r-desc + - r-remotes + - r-withr + license: MIT + license_family: MIT + size: 40945 + timestamp: 1757496123649 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-data.table-1.17.8-r44h1c8cec4_1.conda + sha256: c85f8832e1aafcdb6dcf79c0547696aac91f9b12eda2e765184f3146b9463481 + md5: 3a4754795ca474da5ce390d8510fb674 + depends: + - __glibc >=2.17,<3.0.a0 + - _openmp_mutex >=4.5 + - libgcc >=14 + - libzlib >=1.3.1,<2.0a0 + - r-base >=4.4,<4.5.0a0 + license: MPL-2.0 + license_family: OTHER + size: 2302139 + timestamp: 1757499607287 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-dbi-1.2.3-r44hc72bb7e_2.conda + sha256: a34cf4b5171f62a6dbeb873fe820145af9e47df3fb769b283c8f0797409ec2cc + md5: d69f993e8203a768e65ca0d81fa102ee + depends: + - r-base >=4.4,<4.5.0a0 + license: LGPL-2.1-or-later + license_family: LGPL + size: 873570 + timestamp: 1757467587740 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-dbplyr-2.5.1-r44hc72bb7e_1.conda + sha256: 1040c548f44bbd3e96f2835ab84f23bf6e0820dd434fc304173fd1650f0e847a + md5: a4dba4f5f9bca1b52d1fbd5fa210ad62 + depends: + - r-base >=4.4,<4.5.0a0 + - r-blob >=1.2.0 + - r-cli >=3.6.1 + - r-dbi >=1.1.3 + - r-dplyr >=1.1.2 + - r-glue >=1.6.2 + - r-lifecycle >=1.0.3 + - r-magrittr + - r-pillar >=1.9.0 + - r-purrr >=1.0.1 + - r-r6 >=2.2.2 + - r-rlang >=1.1.1 + - r-tibble >=3.2.1 + - r-tidyr >=1.3.0 + - r-tidyselect >=1.2.1 + - r-vctrs >=0.6.3 + - r-withr >=2.5.0 + license: MIT + license_family: MIT + size: 1218259 + timestamp: 1757537458758 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-desc-1.4.3-r44hc72bb7e_2.conda + sha256: 0b511eadd8299b370b23f45268a9f3de9015457913e86e373fc34184e65c155e + md5: 24fe88af88fc79dd249b6ffdbd32af06 + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli + - r-r6 + - r-rprojroot + license: MIT + license_family: MIT + size: 341014 + timestamp: 1757463156255 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-digest-0.6.38-r44h3697838_0.conda + sha256: 5b45c9a9e9f191ff59bde98394a1bc1382a5850b1c762b0d22b881a8e29ede0f + md5: bf0fe8334f6162e0b18b8b928e0847d0 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL2 + size: 219254 + timestamp: 1762994257421 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-distributional-0.5.0-r44hc72bb7e_1.conda + sha256: 746ef0cf6f1833714422025f4321b17a7cee08a95ed0399883c5318ddafe0d55 + md5: 88b6e55fbf443edb4360bcd18d7b3566 + depends: + - r-base >=4.4,<4.5.0a0 + - r-digest + - r-ellipsis + - r-farver + - r-generics + - r-ggplot2 + - r-lifecycle + - r-numderiv + - r-rlang >=0.4.5 + - r-scales + - r-vctrs >=0.3.0 + license: GPL-3.0-only + license_family: GPL3 + size: 482754 + timestamp: 1757517211757 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-dplyr-1.1.4-r44h3697838_2.conda + sha256: 3d75538e4ea884889d20ddff1e2bd5779a810f377abbb6c9c86f811337219270 + md5: 464c2dea00bfea4dbd3d345ca98dabc0 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-ellipsis + - r-generics + - r-glue >=1.3.2 + - r-lifecycle >=1.0.0 + - r-magrittr >=1.5 + - r-pillar >=1.5.1 + - r-r6 + - r-rlang >=0.4.10 + - r-tibble >=2.1.3 + - r-tidyselect >=1.1.0 + - r-vctrs >=0.3.5 + license: MIT + license_family: MIT + size: 1427096 + timestamp: 1757497362090 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-dtplyr-1.3.2-r44hc72bb7e_1.conda + sha256: bfbfdca008b172a0cb2a65eb7cdc18fc303f12944c61e9f5f9f253320bf857ea + md5: c5439954262f779f890c70c6dae12d49 + depends: + - r-base >=4.4,<4.5.0a0 + - r-crayon + - r-data.table >=1.13.0 + - r-dplyr >=1.0.3 + - r-ellipsis + - r-glue + - r-lifecycle + - r-rlang + - r-tibble + - r-tidyselect + - r-vctrs + license: MIT + license_family: MIT + size: 412323 + timestamp: 1757564956902 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-ellipsis-0.3.2-r44h54b55ab_4.conda + sha256: 2dca1ba67e61f0aa623bb332f850d293387cd66f112f9a16a9a4735b7c4e5ed8 + md5: 3c5f8d5a0e449576ad703da573d77c76 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + - r-rlang >=0.3.0 + license: MIT + license_family: MIT + size: 43967 + timestamp: 1757440944656 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-evaluate-1.0.5-r44hc72bb7e_1.conda + sha256: f7d36f6a5d47c637a08af09878d60559dd13147ef2d725837b3b8e2411c47f0a + md5: 7b03982e4e8e348f02a51e61e6bb9cc2 + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 111297 + timestamp: 1757447669350 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-fansi-1.0.6-r44h54b55ab_2.conda + sha256: 30f241c34331fbbc22c742dee2765cbb16b39e1ef4c3bef765d2aad746c95040 + md5: e5928aaf6b9d1d1f6f41cd30d00df5d3 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL3 + size: 326256 + timestamp: 1757421542638 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-farver-2.1.2-r44h3697838_2.conda + sha256: 4386f196e54365a83cf0df0c2a389cb286e488bb868f96c4d6c0a4cc1bb69c84 + md5: 830031b95e37195f00524cabc98adba4 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 1427819 + timestamp: 1757441247787 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-fastcluster-1.3.0-r44h3697838_1.conda + sha256: d8ab6aace0bdef69a1badb1b7fdb04747c4a826b69d1362e873fd2b22c87bf7c + md5: 2fdc9d046db49bb92362f5a5d300cc33 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: BSD-2-Clause OR GPL-2.0-only + license_family: GPL + size: 198802 + timestamp: 1757622196934 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-fastmap-1.2.0-r44h3697838_2.conda + sha256: 202cbae1f393f944f3465141a78ffce056d3109b21d6437955f4e8a9af8b30e6 + md5: 734e73555e5c0962c40c89e6aa7cf1a8 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 74279 + timestamp: 1757421564828 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-fastmatch-1.1_6-r44h54b55ab_1.conda + sha256: 311cc45e386017ae5cc03c664949727da7d18fb61a18a9e104b88e1219f32847 + md5: a54f9705b1cfd806c58c48d22b46d50b + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-only + license_family: GPL2 + size: 50187 + timestamp: 1757500158344 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-filelock-1.0.3-r44h54b55ab_2.conda + sha256: 7bf1c09b71554b0bc4a1ddaeea0c24790c36983048ca06807b4510b2aa6be5d0 + md5: 7c6896398f42a9657f0ef5d4b65da270 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 33925 + timestamp: 1757575999387 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-findpython-1.0.9-r44hc72bb7e_1.conda + sha256: 5afdfec3bf40fd7ba3cf07d323a9e929b216de95e2f6a675879bb4fbb1c41297 + md5: 46a15cfd4e9d63ec86421b7df8bfd4d6 + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 31048 + timestamp: 1757998590696 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-fontawesome-0.5.3-r44hc72bb7e_1.conda + sha256: efa7167cc694c30895c687c2288b83a77834cffc42782ae0c452dc3f5ca16461 + md5: 1f426d1938c22f0f59407b25026ccdc3 + depends: + - r-base >=4.4,<4.5.0a0 + - r-htmltools >=0.5.1.1 + - r-rlang >=0.4.10 + license: MIT + license_family: MIT + size: 1339066 + timestamp: 1757461226059 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-forcats-1.0.1-r44hc72bb7e_0.conda + sha256: bf480eada922a019476a5702f4d9b94f092fab3fca6b2af96c9345ee1d727d0f + md5: 57a59cda29c0451334f90e9ffaf63635 + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli + - r-ellipsis + - r-glue + - r-lifecycle + - r-magrittr + - r-rlang >=1.0.0 + - r-tibble + - r-withr + license: MIT + license_family: MIT + size: 424528 + timestamp: 1758793460052 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-formatr-1.14-r44hc72bb7e_3.conda + sha256: cb07de1571306945eb26226073f6962c54ea654957e674c77e99fe66fa00c56e + md5: 7dec5b78870748c0c968564b67db9d00 + depends: + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL + size: 166282 + timestamp: 1757545741110 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-fs-1.6.6-r44h3697838_1.conda + sha256: f6c432fdd417fec422984b3e833040ec16ceb00136fc3cfe539ccf27a2a71fe8 + md5: 5e88ccbc8cd62b4f388b9bcb83624861 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 511571 + timestamp: 1757422386782 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-futile.logger-1.4.3-r44hc72bb7e_1007.conda + sha256: de8119ea863c14b58d087177ec924bff39e606eef49707171d342655bb45014d + md5: 64e131c6fd19df6fa19c31fff5f0725e + depends: + - r-base >=4.4,<4.5.0a0 + - r-futile.options + - r-lambda.r >=1.1.0 + license: LGPL-3.0-only + license_family: LGPL + size: 106324 + timestamp: 1757634111319 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-futile.options-1.0.1-r44hc72bb7e_1006.conda + sha256: d560fa200628c6a1df32c077f77224abe6a98a2b1e250b7f6cfe1d301b90b47b + md5: 569ec6607e422c9ad9f622eed1bc55fb + depends: + - r-base >=4.4,<4.5.0a0 + license: LGPL-3.0-only + license_family: LGPL + size: 29552 + timestamp: 1757607816590 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-gargle-1.6.0-r44h785f33e_1.conda + sha256: d6d7b47c1c1dddc70808b5d5e5e2a3490b9725bbf24d61d24d3d59c262acff66 + md5: 46c673c57aa0f8f3e210c11a478ce33c + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli >=3.0.0 + - r-fs >=1.3.1 + - r-glue >=1.3.0 + - r-httr >=1.4.0 + - r-jsonlite + - r-lifecycle + - r-openssl + - r-rappdirs + - r-rlang >=1.0.0 + - r-rstudioapi + - r-withr + license: MIT + license_family: MIT + size: 578608 + timestamp: 1758414065943 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-generics-0.1.4-r44hc72bb7e_1.conda + sha256: 82a8f7ee79ff61f2378dab81b81ecbc7745b01f9be5df78f6f0e073a53f507af + md5: a593d8a24e2d841863085e8c5b50b9fa + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 89321 + timestamp: 1757455977099 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-ggplot2-4.0.1-r44h785f33e_0.conda + sha256: 8ed4fb8da03bab428faabb28c3162ba993b14ecda7b97de70f2695cdfa0dada5 + md5: d8dae97d3a0c3136218789c5d10a71f4 + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli + - r-glue + - r-gtable >=0.3.6 + - r-isoband + - r-lifecycle >=1.0.1 + - r-rlang >=1.1.0 + - r-s7 + - r-scales >=1.4.0 + - r-vctrs >=0.6.0 + - r-withr >=2.5.0 + license: MIT + license_family: MIT + size: 7814775 + timestamp: 1763113271364 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-glue-1.8.0-r44h54b55ab_1.conda + sha256: 2ca92b5f3de0ed821f065fa85c86b4c41275373a9884b81b6799ee1613a0cec2 + md5: 8c0119ef7862a900727cd6845ee1511c + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 165602 + timestamp: 1757421255719 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-googledrive-2.1.2-r44hc72bb7e_1.conda + sha256: 4bc9f5f8b07e3657a6ff0f92d67667484f954b90b802926d7378345e1d668887 + md5: 7973919529f7668fc545c39202832c8b + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli >=3.0.0 + - r-gargle >=1.6.0 + - r-glue >=1.4.2 + - r-httr + - r-jsonlite + - r-lifecycle + - r-magrittr + - r-pillar >=1.9.0 + - r-purrr >=1.0.1 + - r-rlang >=1.0.2 + - r-tibble >=2.0.0 + - r-uuid + - r-vctrs >=0.3.0 + - r-withr + license: MIT + license_family: MIT + size: 1234398 + timestamp: 1758449278742 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-googlesheets4-1.1.2-r44h785f33e_1.conda + sha256: 6aa0bac72b9843709938b2fd43abdfc6a3e4879fabee404d93bf2041b751a26f + md5: 844b415f089066c066986190677f554b + depends: + - r-base >=4.4,<4.5.0a0 + - r-cellranger + - r-cli >=3.0.0 + - r-curl + - r-gargle >=1.2.0 + - r-glue >=1.3.0 + - r-googledrive >=2.0.0 + - r-httr + - r-ids + - r-magrittr + - r-purrr + - r-rematch2 + - r-rlang >=0.4.11 + - r-tibble >=2.1.1 + - r-vctrs >=0.2.3 + license: MIT + license_family: MIT + size: 524627 + timestamp: 1758457266256 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-gplots-3.2.0-r44hc72bb7e_1.conda + sha256: 6c48cac0172d181a71e3c2e40853204cd2642410f26c1457c7a1655c562378e8 + md5: 67c97f6e2ffae37f5e7b1fde0ede1395 + depends: + - r-base >=4.4,<4.5.0a0 + - r-catools + - r-gtools + - r-kernsmooth + license: GPL-2.0-only + license_family: GPL2 + size: 499926 + timestamp: 1757504966893 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-gridextra-2.3-r44hc72bb7e_1007.conda + sha256: 7e26244d122507ea2ef60fb2ac8eef3bb8a133055af63e4200a73fd11c1369d0 + md5: 6203b49d9dc8ea0619a391314759c91b + depends: + - r-base >=4.4,<4.5.0a0 + - r-gtable + license: GPL-2.0-or-later + license_family: GPL3 + size: 1050499 + timestamp: 1757484947406 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-gtable-0.3.6-r44hc72bb7e_1.conda + sha256: e7621253555552d6c29103a8e21c0396712b5de09012263e882bfa0b1e59c812 + md5: 1f5b8f9f0cdb53b843c4e2481e22c8c0 + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli + - r-glue + - r-lifecycle + - r-rlang + license: MIT + license_family: MIT + size: 228683 + timestamp: 1757463528706 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-gtools-3.9.5-r44h54b55ab_2.conda + sha256: 86e0010bb6cf91b1c1c124e63d82271e9a835025a3c4592cd6648b40a0a58448 + md5: 3dc669378e9c37d26cb2616ff9d67243 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-only + license_family: GPL2 + size: 372711 + timestamp: 1757480817023 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-haven-2.5.5-r44h6d565e7_1.conda + sha256: 6f0b2111552ca0f64c7d204d3dfc391e296e67d1e601fa3b34bf16c7d1561663 + md5: f7f18c6c2cfe78556e38fa7e8ef65d7b + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - libzlib >=1.3.1,<2.0a0 + - r-base >=4.4,<4.5.0a0 + - r-cli >=3.0.0 + - r-cpp11 + - r-forcats >=0.2.0 + - r-hms + - r-lifecycle + - r-readr >=0.1.0 + - r-rlang >=0.4.0 + - r-tibble + - r-tidyselect + - r-vctrs >=0.3.0 + license: MIT + license_family: MIT + size: 385319 + timestamp: 1757525036341 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-highr-0.11-r44hc72bb7e_2.conda + sha256: fade9dd156a8045aa2cd0577fdad67ade4eda4df24b38b60e689693793b591c4 + md5: 1d27ac83e555dcd74a3f51961bccd5e8 + depends: + - r-base >=4.4,<4.5.0a0 + - r-xfun >=0.18 + license: GPL-2.0-or-later + license_family: GPL + size: 56962 + timestamp: 1757447790352 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-hms-1.1.4-r44hc72bb7e_0.conda + sha256: 5a016b1533a9dc39c66f61209b0a70979289f18cbeace0377b3dfdb63c69465f + md5: 6bb21cd6a482d5c9bb068368a018e750 + depends: + - r-base >=4.4,<4.5.0a0 + - r-ellipsis + - r-lifecycle + - r-pkgconfig + - r-rlang + - r-vctrs >=0.2.1 + license: MIT + license_family: MIT + size: 112923 + timestamp: 1760687917284 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-htmltools-0.5.8.1-r44h3697838_2.conda + sha256: 81bfeec56e3df8c759c3daa092314eded2c52dc1fbabdf3f429298e1d32bfed1 + md5: 635b9956a4eebdefccb8f2e75b5dfc92 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-base64enc + - r-digest + - r-ellipsis + - r-fastmap >=1.1.0 + - r-rlang >=0.4.10 + license: GPL-2.0-or-later + license_family: GPL3 + size: 367249 + timestamp: 1757453532390 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-httr-1.4.7-r44hc72bb7e_2.conda + sha256: 5ee37625cc31e589414b8c69d618b2143d87b0ef7632867b758c86e2e7f52262 + md5: cad0b6a9f01e28d56bd5f456112f780d + depends: + - r-base >=4.4,<4.5.0a0 + - r-curl >=0.9.1 + - r-jsonlite + - r-mime + - r-openssl >=0.8 + - r-r6 + license: MIT + license_family: MIT + size: 486507 + timestamp: 1758405214312 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-httr2-1.2.1-r44hc72bb7e_1.conda + sha256: 364615086cf2b5c56a9373344d0883ea6bdf5ef353b5337239d38eee2da20661 + md5: 98acf52b1dfb1164d63a5f700b7b1230 + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli >=3.0.0 + - r-curl >=5.1.0 + - r-glue + - r-lifecycle + - r-magrittr + - r-openssl + - r-r6 + - r-rappdirs + - r-rlang >=1.1.0 + - r-vctrs >=0.6.3 + - r-withr + license: MIT + license_family: MIT + size: 778683 + timestamp: 1758405116717 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-hwriter-1.3.2.1-r44hc72bb7e_4.conda + sha256: 4f2e783f85ea1af3a30ef2b028873b0c8835d3b188d1a7cab809ab7f4aa2abb9 + md5: 8c335e4b6376a7fc2214f679248fefa1 + depends: + - r-base >=4.4,<4.5.0a0 + license: LGPL-2.1-only + license_family: LGPL + size: 124225 + timestamp: 1758269018229 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-ids-1.0.1-r44hc72bb7e_5.conda + sha256: d99db73e785fd842519325572cc7bc888fcb3edb6cced2463ee1ff29d9375d16 + md5: 97776ff1ab24b9c717f554c59e6a834a + depends: + - r-base >=4.4,<4.5.0a0 + - r-openssl + - r-uuid + license: MIT + license_family: MIT + size: 130138 + timestamp: 1758407699755 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-igraph-2.1.4-r44hadbbdbc_1.conda + sha256: fee28fc1c6ea874567a9b32635260fb9ebd86be1577163051f782460cbcff85f + md5: 70b528cd2a002c0f169980cec5ece732 + depends: + - __glibc >=2.17,<3.0.a0 + - glpk >=5.0,<6.0a0 + - gmp >=6.3.0,<7.0a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=14 + - libgfortran + - libgfortran5 >=14.3.0 + - liblapack >=3.9.0,<4.0a0 + - liblzma >=5.8.1,<6.0a0 + - libstdcxx >=14 + - libxml2 >=2.13.8,<2.14.0a0 + - libzlib >=1.3.1,<2.0a0 + - r-base >=4.4,<4.5.0a0 + - r-cli + - r-cpp11 >=0.5.0 + - r-lifecycle + - r-magrittr + - r-matrix + - r-pkgconfig >=2.0.0 + - r-rlang + - r-vctrs + license: GPL-2.0-or-later + license_family: GPL3 + size: 5149303 + timestamp: 1758712176795 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-inline-0.3.21-r44hc72bb7e_1.conda + sha256: add6c7334f1b6237959dd3f005e7497db852ade98c8c8d736ebee43da2fb3590 + md5: 45852ac60d22ea9af166fff517d74d13 + depends: + - r-base >=4.4,<4.5.0a0 + license: LGPL-2.0-or-later + license_family: LGPL + size: 139619 + timestamp: 1757514255671 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-isoband-0.2.7-r44h3697838_4.conda + sha256: dabbb6eb4cf067349ff7e696a89ab898fe1571b25120fd2e9cf778fae0ac4b26 + md5: e10cbac3946128930d5da63b32e4baa4 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 1626389 + timestamp: 1757441391901 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-jquerylib-0.1.4-r44hc72bb7e_4.conda + sha256: 95989e7eea19dd3a0c729585a20555388537dc033df120bb7baac26601232f75 + md5: 24ba3d3b0b5d5c19fc549cca0b226388 + depends: + - r-base >=4.4,<4.5.0a0 + - r-htmltools + license: MIT + license_family: MIT + size: 306791 + timestamp: 1757459450146 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-jsonlite-2.0.0-r44h54b55ab_1.conda + sha256: faff2faef7e73c5fd02fea6bbe8aa9e444fabf7f56e8d932c7babc51ee76a290 + md5: 26de9e385370e6bf7d0ca13a3e630d7f + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 639260 + timestamp: 1757419504603 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-kernsmooth-2.23_26-r44ha0a88a1_1.conda + sha256: 71e7c91b230748646fdcae0587a94bc8ea3a5e659e950a0adff17bf6a6007cd3 + md5: 9737719ec08adb623be7358f45221be2 + depends: + - __glibc >=2.17,<3.0.a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=14 + - libgfortran + - libgfortran5 >=14.3.0 + - r-base >=4.4,<4.5.0a0 + license: Unlimited + license_family: Other + size: 101678 + timestamp: 1757457740465 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-knitr-1.50-r44hc72bb7e_1.conda + sha256: 564ad47756e083d0c7893d5c742ffd83f9334afea11398f72f90ee78637b1714 + md5: 3841100c8b6f9beb6291b3498ff3b835 + depends: + - r-base >=4.4,<4.5.0a0 + - r-evaluate >=0.15 + - r-highr >=0.11 + - r-xfun >=0.51 + - r-yaml >=2.1.19 + license: GPL-2.0-or-later + license_family: GPL + size: 1035098 + timestamp: 1757457681904 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-labeling-0.4.3-r44hc72bb7e_2.conda + sha256: e2fe5839dce12c1e7fffdfb83387ae58c93417acb520cd258e8460cc7da533db + md5: 891625b27729556fcac5cb5a3040b10f + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 70089 + timestamp: 1757456017204 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-lambda.r-1.2.4-r44hc72bb7e_5.conda + sha256: 625092566077b110479496f13d610eb7dd128950c5da37b53706f18b6aab39b6 + md5: 52dff70fa7503cd2fff2cdbd48e21b6f + depends: + - r-base >=4.4,<4.5.0a0 + - r-formatr + license: LGPL-3.0-only + license_family: LGPL + size: 121245 + timestamp: 1757608579337 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-languageserver-0.3.16-r44h54b55ab_2.conda + sha256: fc7645900f7243a76ced80d316f6a3bf49e7091152056e568474b7df751d6a58 + md5: 2ab49819a50fae51046c7a89afdbb33f + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + - r-callr >=3.0.0 + - r-collections >=0.3.0 + - r-desc >=1.2.0 + - r-fs >=1.3.1 + - r-jsonlite >=1.6 + - r-lintr >=2.0.0 + - r-r6 >=2.4.1 + - r-repr >=1.1.0 + - r-roxygen2 >=7.0.0 + - r-stringi >=1.1.7 + - r-styler >=1.2.0 + - r-xml2 >=1.2.2 + - r-xmlparsedata >=1.0.3 + license: MIT + license_family: MIT + size: 732454 + timestamp: 1757577903671 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-lattice-0.22_7-r44h54b55ab_1.conda + sha256: f5b32cef2955fd9b5b9b105d62b64bdfdc67ecc231042a2e4b433795570ac045 + md5: a4778b8f282dfa692f0bc55afbc5e31b + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL3 + size: 1383358 + timestamp: 1757424594684 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-lazyeval-0.2.2-r44h54b55ab_6.conda + sha256: 13af0c878a35ee8b337505a6e30198c168d081720dbe4f2d41bd358936447c13 + md5: b4a13a6b0e20b94566b42884a4b423db + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-3.0-only + license_family: GPL3 + size: 161026 + timestamp: 1757422163603 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-lifecycle-1.0.4-r44hc72bb7e_2.conda + sha256: 795a933f29cd2910c9100b0a26e67538f6d981de6eec2fdb30f2060d9eb555c9 + md5: e41e892cef6cad8fb15c6e7af9a19fa0 + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli >=3.4.0 + - r-glue + - r-rlang >=1.0.6 + license: MIT + license_family: GPL3 + size: 124347 + timestamp: 1757452045387 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-lintr-3.2.0-r44hc72bb7e_1.conda + sha256: 2b283581db7167e9f1fc3c8d600e5bd9c1593e8ddb879b9f0a046b13e452d18f + md5: e0bed290853df85fd250aa971f99370a + depends: + - r-backports + - r-base >=4.4,<4.5.0a0 + - r-codetools + - r-crayon + - r-cyclocomp + - r-digest + - r-glue + - r-jsonlite + - r-knitr + - r-rex + - r-xml2 >=1.0.0 + - r-xmlparsedata >=1.0.5 + license: MIT + license_family: MIT + size: 1356871 + timestamp: 1757562142869 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-locfit-1.5_9.12-r44h54b55ab_1.conda + sha256: d0828d0e8e14244b5eac6c2783ffac8f00a1359ac990725e28660d536cbcf6ae + md5: d3c0181f1b8366f20a09c8dc6c33a0bd + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + - r-lattice + license: GPL-2.0-or-later + license_family: GPL + size: 563274 + timestamp: 1757585366385 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-loo-2.8.0-r44hc72bb7e_1.conda + sha256: a5fe8836984d095a0d1cdd27c6ec7e4c8d02490f0d43a59e6f2d7a089601809c + md5: 66a9e0b3bb78e47e5133ae3b57049a02 + depends: + - r-base >=4.4,<4.5.0a0 + - r-checkmate + - r-matrixstats >=0.52 + - r-posterior >=1.5.0 + license: GPL-3.0-or-later + license_family: GPL3 + size: 1782656 + timestamp: 1757548044680 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-lubridate-1.9.4-r44h54b55ab_1.conda + sha256: 58a1ee92c746caa5ce431e5b99ea1bad31d2a18cfa7df104855dc2207c9c53c3 + md5: 51f61a093a8a2f55715740ca3df3e64b + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + - r-generics + - r-timechange >=0.1.1 + license: GPL-2.0-or-later + license_family: GPL2 + size: 984868 + timestamp: 1757500181448 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-magrittr-2.0.4-r44h54b55ab_0.conda + sha256: 7ef6fde1e3be22eba5886aec1c7fc24ea0b0dfe67c3ab0e5d7d32c9a033acebf + md5: d2bddfccf4e15346931d49a61dc9d1ce + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 212139 + timestamp: 1757677983678 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-matrix-1.7_4-r44h0e4624f_1.conda + sha256: 50f3e456ed33a0d6c61800458f4ac010e2ffe7a8850689549cd4f5909d62f738 + md5: bdb8960fd8da7fe82ea3c729ffe29937 + depends: + - __glibc >=2.17,<3.0.a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=14 + - liblapack >=3.9.0,<4.0a0 + - r-base >=4.4,<4.5.0a0 + - r-lattice + license: GPL-2.0-or-later + license_family: GPL3 + size: 4263998 + timestamp: 1757441226105 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-matrixstats-1.5.0-r44h54b55ab_1.conda + sha256: 6ccf1752c744f0a735d8975b4e9b7405ad8746a85b7e059adca1059108490316 + md5: b88ecfe0219a5dbba488de312160f982 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: Artistic-2.0 + license_family: OTHER + size: 485637 + timestamp: 1757442403989 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-memoise-2.0.1-r44hc72bb7e_4.conda + sha256: 67f584820d51e2a121bc7113365733b9d00dee2f270113cb212d1ecfd0e7b038 + md5: eea36c929b57cf85e2a263ccdf265094 + depends: + - r-base >=4.4,<4.5.0a0 + - r-cachem + - r-rlang >=0.4.10 + license: MIT + license_family: MIT + size: 57725 + timestamp: 1757456205848 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-mgcv-1.9_4-r44h0e4624f_0.conda + sha256: 99937b49abea6a327a259b7542e0bafd44bee95e8952ec20a0066badc01afa6c + md5: aefbc031d18739a5b091eaa598b6e59b + depends: + - __glibc >=2.17,<3.0.a0 + - _openmp_mutex >=4.5 + - libblas >=3.9.0,<4.0a0 + - libgcc >=14 + - liblapack >=3.9.0,<4.0a0 + - r-base >=4.4,<4.5.0a0 + - r-matrix + - r-nlme >=3.1_64 + license: GPL-2.0-or-later + license_family: GPL2 + size: 3645604 + timestamp: 1762539580860 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-mime-0.13-r44h54b55ab_1.conda + sha256: 42f450eed2f6b97ff219c8e97885ee982101392f2a5a888e0469d771f74afe6c + md5: d8865c99150c4b51a50292a3c5bc52dd + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL + size: 64976 + timestamp: 1757441391748 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-modelr-0.1.11-r44hc72bb7e_3.conda + sha256: 6629887cab6c95d0e47286e416679f22ba83cba4d50f760ac5130bf20b362200 + md5: 24fd2637122778fd80cbf88104ce1150 + depends: + - r-base >=4.4,<4.5.0a0 + - r-broom + - r-dplyr + - r-magrittr + - r-purrr >=0.2.2 + - r-rlang >=0.2.0 + - r-tibble + - r-tidyr >=0.8.0 + license: GPL-3 + license_family: GPL3 + size: 221137 + timestamp: 1757531792899 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-munsell-0.5.1-r44hc72bb7e_2.conda + sha256: dc7e6deee2782e64e061976cb895f40e95105bd99b297af83ab4a7c33b17fff0 + md5: 79407650271d42d8aadfe756f89521c7 + depends: + - r-base >=4.4,<4.5.0a0 + - r-colorspace + license: MIT + license_family: MIT + size: 247057 + timestamp: 1757455963570 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-nlme-3.1_168-r44heaba542_1.conda + sha256: 92c691b95f39c5fd79fc4e3cffa33495e832bca7a599f9406dbbb7a59d5cd54d + md5: 1d695b83f5865a7398a808739afc6725 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libgfortran + - libgfortran5 >=14.3.0 + - r-base >=4.4,<4.5.0a0 + - r-lattice + license: GPL-2.0-or-later + license_family: GPL3 + size: 2353144 + timestamp: 1757441176456 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-numderiv-2016.8_1.1-r44hc72bb7e_7.conda + sha256: 6634fed10b95f5cad06b6b2a2827f41782a70ca57d8a54671a8b5a36f9a726ec + md5: 123c352894dab867d542ca0de4dc2cb9 + depends: + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-only + license_family: GPL2 + size: 129599 + timestamp: 1757460030284 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-openssl-2.3.4-r44h50f7d53_0.conda + sha256: cb59b197800e6980784b1854eb14a715f70c6ed1f616119a70b9c4fe18e8c770 + md5: ebcdf3884fb953a60e2159df4913bfcb + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - openssl >=3.5.3,<4.0a0 + - r-askpass + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 691865 + timestamp: 1759238508219 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-phangorn-2.12.1-r44hf1899b2_3.conda + sha256: 4c7b861781ac9516c5f609f3de4bcd931688244c2a6dacd57ea9176487c80f79 + md5: c6abdc8c64545b2411e48166d5b76e3a + depends: + - __glibc >=2.17,<3.0.a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=14 + - libstdcxx >=14 + - r-ape >=5.6 + - r-base >=4.4,<4.5.0a0 + - r-digest + - r-fastmatch + - r-generics + - r-igraph >=1.0 + - r-matrix + - r-quadprog + - r-rcpp + license: GPL-2.0-or-later + license_family: GPL3 + size: 2630063 + timestamp: 1758800801542 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-pillar-1.11.1-r44hc72bb7e_0.conda + sha256: 10739851530bc98feff7bf68acd2285dfddc7a8587b33c51a495f014744a3f33 + md5: 89635525b35ad6b443343237b65526ae + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli + - r-crayon >=1.3.4 + - r-ellipsis + - r-fansi + - r-lifecycle + - r-rlang >=0.3.0 + - r-utf8 >=1.1.0 + - r-vctrs >=0.2.0 + license: GPL-3.0-only + license_family: GPL3 + size: 629814 + timestamp: 1758149812405 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-pkgbuild-1.4.8-r44hc72bb7e_1.conda + sha256: 9fdfee8ede6ca0dc94e8f11eeaf7808570a050fbf59de96fb83e09f3133bebd2 + md5: 842576a213f96e7abacdc1a5afc42dc8 + depends: + - r-base >=4.4,<4.5.0a0 + - r-callr >=3.2.0 + - r-cli + - r-crayon + - r-desc + - r-prettyunits + - r-r6 + - r-rprojroot + - r-withr >=2.1.2 + license: GPL-3.0-only + license_family: GPL3 + size: 221541 + timestamp: 1757496269823 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-pkgconfig-2.0.3-r44hc72bb7e_5.conda + sha256: 6845c972210ee39fed0db7326685067cd213fc8d1136e0a2060063f3921ee494 + md5: bdc0f46c7111e1d4f7eb7133374b47b7 + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 27374 + timestamp: 1757447594870 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-pkgload-1.4.1-r44hc72bb7e_0.conda + sha256: ad911229df18867141da929a35f6dede2f5afe86f9dcd80f39d188a5d515b3ee + md5: cf81bb0e096c62c076f43e4b6a08221c + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli >=3.3.0 + - r-desc + - r-fs + - r-glue + - r-lifecycle + - r-pkgbuild + - r-processx + - r-rlang >=1.1.1 + - r-rprojroot + - r-withr >=2.4.3 + license: GPL-3.0-only + license_family: GPL3 + size: 241155 + timestamp: 1758654256202 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-plogr-0.2.0-r44hc72bb7e_1007.conda + sha256: 62e7c9a6ff30cc8a58cc567a3e5caf9183bd085dc1fd4f54150f2c37602e34c8 + md5: db26d93fadddeded9e92aace6229ba6b + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 23011 + timestamp: 1757524759618 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-plyr-1.8.9-r44h3697838_3.conda + sha256: f0b9e9d94a136501a12fa75dc02d5dc86c7d2cfd559bedc9fde78d17cd6d54a4 + md5: d2d7f7c69444244c1c32e577061fd07b + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-rcpp >=0.11.0 + license: MIT + license_family: MIT + size: 787332 + timestamp: 1757441822091 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-png-0.1_8-r44h6b2d295_3.conda + sha256: 093e83831777365cc1f2b8976090f569b9ab5d351c7c3adf4dd0b0f1a61b7e33 + md5: 0be41efd46641a98cb1a3ae062ebaadc + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libpng >=1.6.50,<1.7.0a0 + - libzlib >=1.3.1,<2.0a0 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-only OR GPL-3.0-only + license_family: GPL3 + size: 61134 + timestamp: 1757489834137 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-posterior-1.6.1-r44hc72bb7e_1.conda + sha256: 05d13f4e2d3e63f43cf73a2fa3f8fd464a7096f350002d2dfe732ff53be4e747 + md5: 22f8408e6d053703b1aeda29ef3e8201 + depends: + - r-abind + - r-base >=4.4,<4.5.0a0 + - r-checkmate + - r-distributional + - r-matrixstats + - r-pillar + - r-rlang >=0.4.7 + - r-tensora + - r-tibble >=3.0.0 + - r-vctrs + license: BSD-3-Clause + license_family: BSD + size: 1013692 + timestamp: 1757534127643 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-prettyunits-1.2.0-r44hc72bb7e_2.conda + sha256: cbee3643f2e5b45e52921f5a6246f572cadf5dcf540d352bbc0ff4ad0a96aa4f + md5: d0ddc1e17050afcb5f13ecdc1f4aba92 + depends: + - r-assertthat + - r-base >=4.4,<4.5.0a0 + - r-magrittr + license: MIT + license_family: MIT + size: 161041 + timestamp: 1757463122545 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-processx-3.8.6-r44h54b55ab_1.conda + sha256: adac3d04630ddbe7e3896dad7c921e628291845ea862839e67d0583108bfd802 + md5: 1d5b82b45034841580deb4e0faabc585 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + - r-ps >=1.2.0 + - r-r6 + license: MIT + license_family: MIT + size: 340153 + timestamp: 1757460034899 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-progress-1.2.3-r44hc72bb7e_2.conda + sha256: 23cf5fe12d5344af6dcc74d89504dee70b2534c62dee48b03515e44c27465c45 + md5: a6a93fa6ae11444ea50d897f35db7e5d + depends: + - r-base >=4.4,<4.5.0a0 + - r-crayon + - r-hms + - r-prettyunits + - r-r6 + license: MIT + license_family: MIT + size: 96089 + timestamp: 1757484966329 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-ps-1.9.1-r44h54b55ab_1.conda + sha256: db7845901d09f0ef0124b8220153ea5b005610710a2897214ab130b3bf158342 + md5: 5560766b6a9358409fcd958a3e74bc88 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: BSD-3-Clause + license_family: BSD + size: 407972 + timestamp: 1757421317406 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-purrr-1.2.0-r44h54b55ab_0.conda + sha256: b293d31a211cf05e98e47314bc964bdf7ea81f1a535c682a72daa9e1fb419ddb + md5: 1324935a2dd98149de3638bf288f1843 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + - r-cli >=3.4 + - r-lifecycle >=1.0.3 + - r-magrittr >=1.5 + - r-rlang >=0.4.10 + - r-vctrs >=0.5 + license: MIT + license_family: MIT + size: 547664 + timestamp: 1762265185488 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-quadprog-1.5_8-r44ha0a88a1_7.conda + sha256: 8e212a3db53ca0c06e05afef7d13e2ec50d172e5ab71e724949b9d7bc79d696e + md5: 8c7c2b5421c37596fd61dcd6d906257c + depends: + - __glibc >=2.17,<3.0.a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=14 + - libgfortran + - libgfortran5 >=14.3.0 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL3 + size: 48653 + timestamp: 1757458190140 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-quickjsr-1.8.1-r44h3697838_0.conda + sha256: 0e40d276c83dcef968fe7691214f8e818ef719d8ef6784ac01227a98c2d38ca4 + md5: 531cdc1587bc655c130d3c8d34509ef5 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 538583 + timestamp: 1758375535975 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-r.cache-0.17.0-r44hc72bb7e_1.conda + sha256: c4afb96d1aabc0b8919fe82c0027ea2c161d2e5b3c2d317702f46826bf84bf39 + md5: 7929ed036c43b12ef5ad41d7ffb0709e + depends: + - r-base >=4.4,<4.5.0a0 + - r-digest >=0.6.13 + - r-r.methodss3 >=1.7.1 + - r-r.oo >=1.23.0 + - r-r.utils >=2.8.0 + license: LGPL-2.1-or-later + license_family: LGPL + size: 128628 + timestamp: 1757498827222 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-r.methodss3-1.8.2-r44hc72bb7e_4.conda + sha256: 01d6e465149cba1ed9aff04bea9c89c355fb6f0e6ba9c83e2f7dd8a8e98795b8 + md5: 146b1ea9d642719c5abb6403844c98eb + depends: + - r-base >=4.4,<4.5.0a0 + license: LGPL-2.1-or-later + license_family: LGPL + size: 99522 + timestamp: 1757447828703 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-r.oo-1.27.1-r44hc72bb7e_1.conda + sha256: ace9a06b9a1784914c71316ab92d8ad3e097bd943f2f4b045c99cad9f92e69f2 + md5: c1b2572ba74dd738c5dd6e363ec84392 + depends: + - r-base >=4.4,<4.5.0a0 + - r-r.methodss3 >=1.7.1 + license: LGPL-2.1-or-later + license_family: LGPL + size: 995636 + timestamp: 1757455955521 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-r.utils-2.13.0-r44hc72bb7e_1.conda + sha256: 2cad4b1b67eec0ee360db33d45c5268a84c6866500a13c2349bb18d9aa01dfba + md5: 9f0e2df8970696fe83a51560e903a013 + depends: + - r-base >=4.4,<4.5.0a0 + - r-r.methodss3 >=1.8.0 + - r-r.oo >=1.23.0 + license: LGPL-2.1-or-later + license_family: LGPL + size: 1428300 + timestamp: 1757484792658 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-r6-2.6.1-r44hc72bb7e_1.conda + sha256: 6cd12790c46f54c67dba4a08961c69cc82f025f2ba7a30f8fae3940d81def8bd + md5: a6809636e421b99a22d374a752c1a2d6 + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 94729 + timestamp: 1757447588692 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-ragg-1.5.0-r44h9f1dc4d_1.conda + sha256: 9c57e4e15a82d314ce391c5463e73e118292147a980dacd801c7e4b52c8ba8e1 + md5: ab5f54efb727eb810de80edda9799895 + depends: + - __glibc >=2.17,<3.0.a0 + - libfreetype >=2.14.0 + - libfreetype6 >=2.14.0 + - libgcc >=14 + - libjpeg-turbo >=3.1.0,<4.0a0 + - libpng >=1.6.50,<1.7.0a0 + - libstdcxx >=14 + - libtiff >=4.7.0,<4.8.0a0 + - libzlib >=1.3.1,<2.0a0 + - r-base >=4.4,<4.5.0a0 + - r-systemfonts >=1.0.3 + - r-textshaping >=0.3.0 + license: MIT + license_family: MIT + size: 591737 + timestamp: 1757532810414 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rappdirs-0.3.3-r44h54b55ab_4.conda + sha256: 81f521fc64831a2444f821f63fd984c57cb6f0fdb97eeb8e29c446b1ca5d763e + md5: 35a2221656a051a8fe4bcfeedda813f2 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 53720 + timestamp: 1757441487351 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rcolorbrewer-1.1_3-r44h785f33e_4.conda + sha256: 7fe0b7282c290b8267d093cdfb4ae093d4a5fdf65f97fdac1d6db998bb96bc32 + md5: bbaae25579d5cea71278441c0b61fde3 + depends: + - r-base >=4.4,<4.5.0a0 + license: Apache-2.0 + license_family: APACHE + size: 67302 + timestamp: 1757452497965 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rcpp-1.1.0-r44h3697838_1.conda + sha256: 9e0414ff18dc62d10769c034f535559736a0b3bd9b7c09f73b4078d3e2d6ee4a + md5: d463481439ac0563a053a1fafd8ce28b + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL2 + size: 2077075 + timestamp: 1757427784298 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rcpparmadillo-15.0.2_2-r44h3704496_0.conda + sha256: 0f8b8db65399e7c94c6d001cae5342947a094b79bebda84da7a9fe483ef8df34 + md5: f226ec583ab05c0b5f54c0296b78862c + depends: + - __glibc >=2.17,<3.0.a0 + - _openmp_mutex >=4.5 + - libblas >=3.9.0,<4.0a0 + - libgcc >=14 + - liblapack >=3.9.0,<4.0a0 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-rcpp >=0.11.0 + license: GPL-2.0-or-later + license_family: GPL2 + size: 1071176 + timestamp: 1758277402161 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rcppeigen-0.3.4.0.2-r44h3704496_1.conda + sha256: 82d1f91325a04e27ad9251d1c11b76f6594efc17be65ee2ebb0a70aff938ebde + md5: 52a1830d3f7f5673e1b60dd0879f8f32 + depends: + - __glibc >=2.17,<3.0.a0 + - libblas >=3.9.0,<4.0a0 + - libgcc >=14 + - liblapack >=3.9.0,<4.0a0 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-matrix >=1.1_0 + - r-rcpp >=0.11.0 + license: GPL-2.0-or-later + license_family: GPL2 + size: 1496550 + timestamp: 1757496080506 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rcppparallel-5.1.11_1-r44hbd9b9cf_2.conda + sha256: cf13c218f4421b29b6cc62121bc9d949fff90e719fc7e413ce19b423ff40cc00 + md5: 2ad23fbaa794dbae9d6150243fcbb60f + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - tbb >=2021.13.0 + - tbb-devel + license: GPL-3.0-or-later + license_family: GPL3 + size: 346686 + timestamp: 1759226967384 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rcurl-1.98_1.17-r44hb79926e_1.conda + sha256: 6c982b031b5c36f941f5f64cfa1ff9009bad98fab13f969fae1579b28e33c5d5 + md5: 9bd68ccae4884d9544d6902be3dfa3d5 + depends: + - __glibc >=2.17,<3.0.a0 + - libcurl >=8.14.1,<9.0a0 + - libgcc >=14 + - liblzma >=5.8.1,<6.0a0 + - libxml2 >=2.13.8,<2.14.0a0 + - libzlib >=1.3.1,<2.0a0 + - r-base >=4.4,<4.5.0a0 + - r-bitops + license: BSD-3-Clause + license_family: BSD + size: 834897 + timestamp: 1757494172816 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-readr-2.1.5-r44h3697838_2.conda + sha256: 18084fac48cac972598bb931a80d008e57dd2bd11ef7b4772aad6857f8760cb3 + md5: 91787b1aa532729d6a80a1f2c1273041 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-cli + - r-clipr + - r-cpp11 + - r-crayon + - r-hms >=0.4.1 + - r-lifecycle >=0.2.0 + - r-r6 + - r-rlang + - r-tibble + - r-tzdb >=0.1.1 + - r-vroom >=1.5.4 + license: MIT + license_family: MIT + size: 845529 + timestamp: 1757517614115 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-readxl-1.4.5-r44h10e25cc_1.conda + sha256: 22b6799c3f56cb49d92a4f8324176768fca4fea7b84642a5fd4a5d059d85c549 + md5: a24384a62f0a25710dc17d232bfe05bc + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libiconv >=1.18,<2.0a0 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-cellranger + - r-cpp11 >=0.4.0 + - r-progress + - r-tibble >=2.0.1 + license: MIT + license_family: MIT + size: 371441 + timestamp: 1757525755711 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rematch-2.0.0-r44hc72bb7e_2.conda + sha256: 651e5cfc5a8116899fd8ce87594601b7a75df82eb790e20cc4af526687c2a1db + md5: 9f6f463b456d1f8c99308fc1022f900a + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 25895 + timestamp: 1757488149590 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rematch2-2.1.2-r44hc72bb7e_5.conda + sha256: dbce30e4ef2ffae2bb3cff19ca247cac23a863e7c7db7ca407d6c25552e56944 + md5: 9c13ea9c0d41379ad6deee9e0c3932e9 + depends: + - r-base >=4.4,<4.5.0a0 + - r-tibble + license: MIT + license_family: MIT + size: 56015 + timestamp: 1757496162089 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-remotes-2.5.0-r44hc72bb7e_2.conda + sha256: 1f4d85feeeab79048cde675287b65e74ee56f0f1df3e9ed537649a735634e28d + md5: af39b8f71e6979bd00e9ef71e40bed1e + depends: + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL3 + size: 436897 + timestamp: 1757451587405 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-repr-1.1.7-r44h785f33e_2.conda + sha256: d3042031f593eef9d930f96358fac052969605c261cfebe4921b2510cabeb6a5 + md5: 5d70ec0db6038ea5a9cb54d9fd66fb2f + depends: + - r-base >=4.4,<4.5.0a0 + - r-base64enc + - r-htmltools + - r-jsonlite + - r-pillar >=1.4.0 + license: GPL-3.0-only + license_family: GPL3 + size: 147720 + timestamp: 1757481827811 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-reprex-2.1.1-r44hc72bb7e_2.conda + sha256: d6aeaa84694fb052ddd71d3bb1a00a95dbc6a4bb6b48ee7d7b52ff9a2fc198bf + md5: e24bcec7d3b3b07baa20f27bbf36a02b + depends: + - pandoc >=2.0 + - r-base >=4.4,<4.5.0a0 + - r-callr >=3.6.0 + - r-cli >=3.2.0 + - r-clipr >=0.4.0 + - r-fs + - r-glue + - r-knitr >=1.23 + - r-lifecycle + - r-rlang >=1.0.0 + - r-rmarkdown + - r-rstudioapi + - r-withr >=2.3.0 + license: MIT + license_family: MIT + size: 502143 + timestamp: 1757575283183 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-reshape2-1.4.5-r44h3697838_0.conda + sha256: b907ee3a5e04da8d424f9b273e3d024418227d08940bd2de158ccf82a83ba758 + md5: 089218ddb8b04cabbd53d7e6f02a785b + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-plyr >=1.8.1 + - r-rcpp + - r-stringr + license: MIT + license_family: MIT + size: 128214 + timestamp: 1762975641310 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/r-restfulr-0.0.16-r44h5ef9028_0.tar.bz2 + sha256: ded27a112169cfaee4c5fb82b8ae9a7f702308be5e6049eb72bfd676c966b0be + md5: 5a366eab5e0c57f1660798e6b1f601d5 + depends: + - bioconductor-s4vectors >=0.13.15 + - bioconductor-s4vectors >=0.44.0,<0.45.0a0 + - libgcc >=13 + - r-base >=4.4,<4.5.0a0 + - r-rcurl + - r-rjson + - r-xml + - r-yaml + license: Artistic-2.0 + license_family: OTHER + size: 448692 + timestamp: 1751055738496 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rex-1.2.1-r44hc72bb7e_4.conda + sha256: 9259ed67be5c7c8d5c800e8ef17e4389ba9e1b5f8427690229ea1f29a7b07383 + md5: a16b5e3814b9705071938a149d710b74 + depends: + - r-base >=4.4,<4.5.0a0 + - r-lazyeval + - r-magrittr + license: MIT + license_family: MIT + size: 125570 + timestamp: 1757451630787 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rjson-0.2.23-r44h3697838_1.conda + sha256: 4bb2b5a567e8523c0a6caf664b7b88ea7cb03591e106bcafbf201f69f2d15ccf + md5: 76fc8d84787c65659696bc6aa1b495e6 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-only + license_family: GPL2 + size: 118914 + timestamp: 1757542200757 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rlang-1.1.6-r44h3697838_1.conda + sha256: fb0713478ade48e0c15103485968b2974563d3b2dd4895f0be4c7970f65c61d9 + md5: dc3e0d136136d2bd2e7f7e1ad53cc408 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-3.0-only + license_family: GPL3 + size: 1569324 + timestamp: 1757420889183 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rmarkdown-2.30-r44hc72bb7e_0.conda + sha256: 6aef124899807471fdc9244fbb8e8da8da4a49ffcb6391f8d064658038ffbb06 + md5: 5ca425ff851063103eb4eb029ac3485f + depends: + - pandoc >=1.14 + - r-base >=4.4,<4.5.0a0 + - r-bslib >=0.2.5.1 + - r-evaluate >=0.13 + - r-fontawesome >=0.5.0 + - r-htmltools >=0.5.1 + - r-jquerylib + - r-jsonlite + - r-knitr >=1.43 + - r-tinytex >=0.31 + - r-xfun >=0.36 + - r-yaml >=2.1.19 + license: GPL-3.0-only + license_family: GPL3 + size: 2086844 + timestamp: 1759149973237 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-roxygen2-7.3.3-r44h3697838_1.conda + sha256: 4f1159b7e3be889ab0a436c8c8d10864dd6f49f0583df5d1cd9b483f18f11ec4 + md5: abe1c315d85b430bf9e632ff87e924d9 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-brew + - r-commonmark + - r-cpp11 + - r-desc >=1.2.0 + - r-digest + - r-knitr + - r-pkgload >=1.0.2 + - r-purrr >=0.3.3 + - r-r6 >=2.1.2 + - r-rlang + - r-stringi + - r-stringr >=1.0.0 + - r-xml2 + license: MIT + license_family: GPL3 + size: 705732 + timestamp: 1757563761505 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rprojroot-2.1.1-r44hc72bb7e_1.conda + sha256: 23db8cfaf9c46b48fe8969c716887972f2d243e60268cca6999fdf1402b89f87 + md5: 51b5479ff88c1260e80ec9195189be13 + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 116982 + timestamp: 1757447577256 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rsqlite-2.4.4-r44h3697838_0.conda + sha256: 1d5fe48e1c3f2ad822061d424558f15fdf5c186c12e146c35c1394f7177530b0 + md5: ea11b6032bad82e5f27fc295a209f0e9 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-bit64 + - r-blob >=1.2.0 + - r-cpp11 + - r-dbi >=1.1.0 + - r-memoise + - r-pkgconfig + - r-plogr >=0.2.0 + license: LGPL-2.1-or-later + license_family: LGPL + size: 1310391 + timestamp: 1762773311174 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-rstan-2.32.7-r44h3697838_0.conda + sha256: fb43fa6442e2e25c44dd9cbfa20b955b0cf81311ba3811d69eab2d6638ed273c + md5: a24abb052bdb58ecbb6f4833346433f2 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-ggplot2 >=3.3.5 + - r-gridextra >=2.3 + - r-inline >=0.3.19 + - r-loo >=2.4.1 + - r-pkgbuild >=1.2.0 + - r-quickjsr + - r-rcpp >=1.0.7 + - r-rcppeigen >=0.3.4.0.0 + - r-rcppparallel >=5.1.4 + - r-stanheaders >=2.32.0 + license: GPL-3.0-or-later + license_family: GPL3 + size: 2389528 + timestamp: 1753958500920 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rstudioapi-0.17.1-r44hc72bb7e_1.conda + sha256: 680512a8a0f6756596269ce388f4c13aa9c63ddb7031692a905460bb6da38546 + md5: a0a0be6415cb5f1eb693c06b8804202f + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 318219 + timestamp: 1757460047965 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-rvest-1.0.5-r44hc72bb7e_1.conda + sha256: 526d48d789ead27e8d3db1cea63499b85ff5ab60652236cac93a83b86788ca27 + md5: 4e10d6e0cab403bd6b19f34e3180e124 + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli + - r-glue + - r-httr >=0.5 + - r-lifecycle >=1.0.0 + - r-magrittr + - r-rlang >=1.0.0 + - r-selectr + - r-tibble + - r-withr + - r-xml2 >=1.3 + license: MIT + license_family: MIT + size: 305335 + timestamp: 1758412932984 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-s7-0.2.0-r44h54b55ab_1.conda + sha256: f0a63122ed35e94194b7c871ab90eb02c5a1feb23b1aad71528a018e6b3a4d44 + md5: 966f2cf726c21c34ad8532c1d9272a55 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 310886 + timestamp: 1757526985173 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-sass-0.4.10-r44h3697838_1.conda + sha256: 5aa378f547b9d77810f2c466a6a2715809e78af700e078d04ad839eb1f2f6e39 + md5: 523ee1e398b97b5369e9f457c6fabf45 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-digest + - r-fs + - r-htmltools + - r-r6 + - r-rappdirs + - r-rlang + license: MIT + license_family: MIT + size: 2316643 + timestamp: 1757464686244 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-scales-1.4.0-r44hc72bb7e_1.conda + sha256: 4e28620ef9c4777bb165f87404466652baac788ccf40904d148ebd27430baa4f + md5: d5a3c79508fd7f6be657f69119265398 + depends: + - r-base >=4.4,<4.5.0a0 + - r-farver >=2.0.0 + - r-labeling + - r-lifecycle + - r-munsell >=0.5 + - r-r6 + - r-rcolorbrewer + - r-viridislite + license: MIT + license_family: MIT + size: 780906 + timestamp: 1757487429584 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-selectr-0.4_2-r44hc72bb7e_5.conda + sha256: 3078d92daa08c2de51979278adcd6677c92cc761576f1ee0fd8cb7b4a1a01194 + md5: 7afd352c040a5a5062277adf3d1773ac + depends: + - r-base >=4.4,<4.5.0a0 + - r-r6 + - r-stringr + license: BSD-3-Clause + license_family: BSD + size: 422926 + timestamp: 1757496800044 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-sm-2.2_6.0-r44heaba542_2.conda + sha256: 6d0bf11b3a276cb4af5d12c25085bbbf66976fefd14175600722db61449102b3 + md5: ee95019837c35b186b930f397b0122f4 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libgfortran + - libgfortran5 >=14.3.0 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL2 + size: 816813 + timestamp: 1757805115696 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-snow-0.4_4-r44hc72bb7e_4.conda + sha256: eac54b6b8dc5470710d289298214679bc2b93cc1041c8e9660c54dd8f30ba8e6 + md5: 8ac3bc039d3753f112dd0a3b969efc32 + depends: + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL + size: 116633 + timestamp: 1757568498963 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-stanheaders-2.32.10-r44ha36cffa_2.conda + sha256: af1f26066258f22242d6d3854396d55e10a93c692476c8a1683bca0560d8b447 + md5: 01866afed48bc5ae9d2bb9c32c15e38e + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libgfortran + - libgfortran5 >=14.3.0 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-rcppeigen >=0.3.4.0.0 + - r-rcppparallel >=5.1.4 + license: BSD-3-Clause + license_family: BSD + size: 1539422 + timestamp: 1759235109702 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-statmod-1.5.1-r44hb1d0f04_0.conda + sha256: 3f49c89bd3ed8773e2dd2c8b5ab71127149d5239884a4f2b466fb46e93bfb08b + md5: 9f4312492e7a626b219c438b0e5ed709 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libgfortran + - libgfortran5 >=14.3.0 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL3 + size: 321254 + timestamp: 1760005102520 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-stringi-1.8.7-r44h2dae267_1.conda + sha256: b813af09921f55ecbcb801eda95a1fd4b6906464f4106423106debe4e51e8da2 + md5: 7a4d6615664ef975115ba883a4846a63 + depends: + - __glibc >=2.17,<3.0.a0 + - icu >=75.1,<76.0a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: FOSS + license_family: OTHER + size: 940213 + timestamp: 1757417790324 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-stringr-1.6.0-r44h785f33e_0.conda + sha256: 257a890269425bc338ee660fba694d28e9cc48e7bea9e6f1af9334123d97965c + md5: 4186ef1048a6a5e5f55c359eb442093f + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli + - r-glue >=1.6.1 + - r-lifecycle >=1.0.3 + - r-magrittr + - r-rlang >=1.0.0 + - r-stringi >=1.5.3 + - r-vctrs + license: MIT + license_family: MIT + size: 318920 + timestamp: 1762269717092 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-styler-1.11.0-r44hc72bb7e_0.conda + sha256: da8c06c9d1d73414323ca836311507325ec6cdad2b1a9c68dc22ec61663edad5 + md5: 6598c21500ebd827212c4380d95bb365 + depends: + - r-backports >=1.1.0 + - r-base >=4.4,<4.5.0a0 + - r-cli >=1.1.0 + - r-magrittr >=2.0.0 + - r-purrr >=0.2.3 + - r-r.cache >=0.14.0 + - r-rematch2 >=2.0.1 + - r-rlang >=0.1.1 + - r-rprojroot >=1.1 + - r-tibble >=1.4.2 + - r-withr >=1.0.0 + - r-xfun >=0.1 + license: MIT + license_family: MIT + size: 809138 + timestamp: 1760338512577 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-survival-3.8_3-r44h54b55ab_1.conda + sha256: 2df0e6ae3936c9b3b7a7fe856a9ef657246c87c5b4ed68aa9703557b661bc821 + md5: 9caffbc2b318c14a97d04f2477b16568 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + - r-matrix + license: LGPL-2.0-or-later + license_family: LGPL + size: 8275876 + timestamp: 1757461467290 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-sys-3.4.3-r44h54b55ab_1.conda + sha256: fb66890ea18fb2cc36ed1349cda0d5da831cdb186344ee7a781adc18ad29ffe1 + md5: 52483832668bfe9a097ce5c438751fee + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 50068 + timestamp: 1757441775859 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-systemfonts-1.3.1-r44h74f4acd_0.conda + sha256: a486bfefb2c34c35939fd2858d427de16ad62dc69545d05405a22b4b30ccef17 + md5: 9dd85dd986c86856fe2b880705fef832 + depends: + - __glibc >=2.17,<3.0.a0 + - libfreetype >=2.14.1 + - libfreetype6 >=2.14.1 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-base64enc + - r-cpp11 >=0.2.1 + - r-jsonlite + - r-lifecycle + license: MIT + license_family: MIT + size: 708059 + timestamp: 1759353223456 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-tensora-0.36.2.1-r44h54b55ab_2.conda + sha256: a2699fe85d5206700753213c9e2870932ecfb332913515434f241b20c00af7a0 + md5: e200aed87bdbeeb5deada9424dacdcd3 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL3 + size: 239439 + timestamp: 1757512562925 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-textshaping-1.0.4-r44h74f4acd_0.conda + sha256: 6af878e7ca8bd127a34cf99cb6c777493d1bcc08c52e056ea5e71b06858b94ab + md5: daafb92653bf443e199e79293f772e5f + depends: + - __glibc >=2.17,<3.0.a0 + - fribidi >=1.0.16,<2.0a0 + - harfbuzz >=12.1.0 + - libfreetype >=2.14.1 + - libfreetype6 >=2.14.1 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-cpp11 >=0.2.1 + - r-lifecycle + - r-stringi + - r-systemfonts >=1.3.0 + license: MIT + license_family: MIT + size: 187772 + timestamp: 1760176887348 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-tibble-3.3.0-r44h54b55ab_1.conda + sha256: 1904f7f80e90ebde1bc138906c9dc15660cd72b65cc0a5986ee5d3d3a728f562 + md5: 0066c02fd1b53ab8020a3899d469f62f + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + - r-fansi >=0.4.0 + - r-lifecycle >=1.0.0 + - r-magrittr + - r-pillar >=1.8.1 + - r-pkgconfig + - r-rlang >=1.0.2 + - r-vctrs >=0.4.2 + license: MIT + license_family: MIT + size: 619014 + timestamp: 1757483188828 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-tidyr-1.3.1-r44h3697838_2.conda + sha256: 4d1bc9a47e3cf1790ed2af3587c22b3b021b4b84da9d7b7896cf3286a712b96a + md5: df33e66a25d160cc01aa3d3dd357fbfb + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-cli >=3.4.1 + - r-dplyr >=1.0.10 + - r-glue + - r-lifecycle >=1.0.3 + - r-magrittr + - r-purrr >=1.0.1 + - r-rlang >=1.0.4 + - r-stringr >=1.5.0 + - r-tibble >=2.1.1 + - r-tidyselect >=1.2.0 + - r-vctrs >=0.5.2 + license: MIT + license_family: MIT + size: 1127563 + timestamp: 1757505592576 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-tidyselect-1.2.1-r44hc72bb7e_2.conda + sha256: 11958a4c8f9e414aa6d7a9d5fc1161423258fce9ab9a19f23f8957dc83688470 + md5: d51e16d0d45d1f5a21f74a9562ffd37d + depends: + - r-base >=4.4,<4.5.0a0 + - r-cli >=3.3.0 + - r-glue >=1.3.0 + - r-lifecycle >=1.0.3 + - r-rlang >=1.0.4 + - r-vctrs >=0.5.2 + - r-withr + license: MIT + license_family: MIT + size: 219708 + timestamp: 1757475755088 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-tidyverse-2.0.0-r44h785f33e_3.conda + sha256: ca71b4873adcbe850c9931f5ce34ca0aa688b9d722d7bb1dbb5bf5c3147131b1 + md5: 6dd18be090bd4e5fc166e049e0e67871 + depends: + - r-base >=4.4,<4.5.0a0 + - r-broom >=1.0.3 + - r-cli >=3.6.0 + - r-conflicted >=1.2.0 + - r-dbplyr >=2.3.0 + - r-dplyr >=1.1.0 + - r-dtplyr >=1.2.2 + - r-forcats >=1.0.0 + - r-ggplot2 >=3.4.1 + - r-googledrive >=2.0.0 + - r-googlesheets4 >=1.0.1 + - r-haven >=2.5.1 + - r-hms >=1.1.2 + - r-httr >=1.4.4 + - r-jsonlite >=1.8.4 + - r-lubridate >=1.9.2 + - r-magrittr >=2.0.3 + - r-modelr >=0.1.10 + - r-pillar >=1.8.1 + - r-purrr >=1.0.1 + - r-ragg >=1.2.5 + - r-readr >=2.1.4 + - r-readxl >=1.4.2 + - r-reprex >=2.0.2 + - r-rlang >=1.0.6 + - r-rstudioapi >=0.14 + - r-rvest >=1.0.3 + - r-stringr >=1.5.0 + - r-tibble >=3.1.8 + - r-tidyr >=1.3.0 + - r-xml2 >=1.3.3 + license: MIT + license_family: MIT + size: 426594 + timestamp: 1758467296446 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-timechange-0.3.0-r44h3697838_2.conda + sha256: 36302fafe88aac4ad676cbee6bd6924b2d534aa18ea62904f29ad14239e58601 + md5: c4ebce9e9efd230e39590177e4707ffd + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-cpp11 >=0.2.7 + license: GPL-3.0-only AND Apache-2.0 + license_family: GPL3 + size: 194780 + timestamp: 1757491240245 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-tinytex-0.57-r44hc72bb7e_1.conda + sha256: 73566e3a6ba4681ec50f157f890ba0c1782ea97851ab310eddc6fd7d156ec460 + md5: 8b64c8bc21ac59c74646dbeb24301ab6 + depends: + - r-base >=4.4,<4.5.0a0 + - r-xfun >=0.5 + license: MIT + license_family: MIT + size: 153241 + timestamp: 1757456334205 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-tzdb-0.5.0-r44h3697838_2.conda + sha256: 4640ae24c27ce009406b392c739d9ef6015ed6243fa3f6b8b466ccc47cf9ad9f + md5: 22de54c4ce316337baf45289e63ab601 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-cpp11 >=0.5.2 + license: MIT + license_family: MIT + size: 554579 + timestamp: 1757489995373 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-utf8-1.2.6-r44h54b55ab_1.conda + sha256: 8baeb9d46c17ce6578155af0f2dce665267eb8a16fa263659ad33e5005c3ed3f + md5: c7bce6987e0f5aaf9bdf5296dd3693c4 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: Apache-2.0 + license_family: APACHE + size: 147724 + timestamp: 1757424747047 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-uuid-1.2_1-r44h54b55ab_1.conda + sha256: 61289d9f8c36c94bc82bb880bf1aa39d8eb7d5bf67a16677947b34e8df58a7e1 + md5: 7cd30b76e660299de48282b6041431c2 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 56491 + timestamp: 1757439879408 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-vctrs-0.6.5-r44h3697838_2.conda + sha256: 3c6522448a44c373a6082fe09737fcd54803044eb533b2b49e8d508eaaa9626a + md5: e93d96165f866807b682452f3c31b1ea + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-cli >=3.4.0 + - r-glue + - r-lifecycle >=1.0.3 + - r-rlang >=1.0.6 + license: MIT + license_family: MIT + size: 1266235 + timestamp: 1757458092486 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-vioplot-0.5.1-r44hc72bb7e_1.conda + sha256: 02ab320c0acce8233a53242d86e0a7b5eca144dcf43f836fb2dd335628dedee2 + md5: fa2cdde0af4a471beca54f4f6e01a17a + depends: + - r-base >=4.4,<4.5.0a0 + - r-sm + - r-zoo + license: BSD-3-Clause + license_family: BSD + size: 359496 + timestamp: 1757963268229 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-viridislite-0.4.2-r44hc72bb7e_3.conda + sha256: 58dcf8f808735a78fdd488ea92471c85245f4ccf81d28903f5fcc4a7c0a63fb9 + md5: f7988a8606e974bba5a4d933236a6f2f + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 1304808 + timestamp: 1757455822836 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-vroom-1.6.6-r44h3697838_0.conda + sha256: 0a9ee50ec2b212b0d9938495b15cfd354d52a64b3204a4f8a9ec176b358aaf63 + md5: 0f28463c6576cbfd53432dd19959ecba + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + - r-bit64 + - r-cli + - r-cpp11 >=0.2.0 + - r-crayon + - r-glue + - r-hms + - r-lifecycle + - r-progress >=1.2.1 + - r-rlang >=0.4.2 + - r-tibble >=2.0.0 + - r-tidyselect + - r-tzdb >=0.1.1 + - r-vctrs >=0.2.0 + - r-withr + license: MIT + license_family: MIT + size: 887431 + timestamp: 1758291277053 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-withr-3.0.2-r44hc72bb7e_1.conda + sha256: dda58385230d6969a184022bbd3d692007d696c3b4491f70ecaa94ee0c5e7dac + md5: 0fdd8ccde2641093c651e2b6a25d85e1 + depends: + - r-base >=4.4,<4.5.0a0 + license: GPL-2.0-or-later + license_family: GPL2 + size: 235008 + timestamp: 1757447310876 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-xfun-0.54-r44h3697838_0.conda + sha256: 98cff801a52bb250c2511cc83d62a336bf3dafe167e1fb31c0b864b9ae5f159c + md5: 9e8703f87d8888d6ef9c9154561d842c + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 584762 + timestamp: 1761852477666 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-xml-3.99_0.17-r44h7c9d5c0_3.conda + sha256: 6e950d297818938f0d0f61ccdb60c2d0f99e733ecbd3e8005d65edad123f91df + md5: 1fb2d58ef329162ad0592bc2eedf4c3e + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - liblzma >=5.8.1,<6.0a0 + - libxml2 >=2.13.8,<2.14.0a0 + - libzlib >=1.3.1,<2.0a0 + - r-base >=4.4,<4.5.0a0 + license: BSD-2-Clause + license_family: BSD + size: 1770911 + timestamp: 1757617358051 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-xml2-1.4.0-r44hc6fd541_1.conda + sha256: f391b6989d7f2939b2060cef78e04785a6a15b855413029d18ab0f3cfb08b99f + md5: 0e2dac513c802712a6d01dc52e54df07 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - liblzma >=5.8.1,<6.0a0 + - libstdcxx >=14 + - libxml2 >=2.13.8,<2.14.0a0 + - libzlib >=1.3.1,<2.0a0 + - r-base >=4.4,<4.5.0a0 + - r-cli + - r-rlang >=1.1.0 + license: GPL-2.0-or-later + license_family: GPL2 + size: 354483 + timestamp: 1757555189504 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-xmlparsedata-1.0.5-r44hc72bb7e_4.conda + sha256: e1cfd008864bfa8635c0cd26ebe301b57041acfaaa060b58ca1cdc55b1d3e9a9 + md5: 4a0ed3cdc553f7e97003d7b5abfd45b4 + depends: + - r-base >=4.4,<4.5.0a0 + license: MIT + license_family: MIT + size: 28848 + timestamp: 1757451920883 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/r-xtable-1.8_4-r44hc72bb7e_7.conda + sha256: b3db7745d076860ca48170a27a4f65448c20fa2af1ead391f531337b06f79d7d + md5: cd78e7159529d6e37c3945e6aad05c2c + depends: + - r-base >=4.4,<4.5.0a0 + license: GPL (>= 2) + license_family: GPL3 + size: 703893 + timestamp: 1757456431084 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-yaml-2.3.10-r44h54b55ab_1.conda + sha256: 6c326208d1337b75d6d8c54034d755f4a3c117d3776fb8b9743b33f1574af47d + md5: cee55143ddb0399c9beb063f0a0b7de1 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + license: BSD-3-Clause + license_family: BSD + size: 120229 + timestamp: 1757421960299 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/r-zoo-1.8_14-r44h54b55ab_1.conda + sha256: bdd95f2f0070220dbeee0ca9dfbd36de8c232d60121e0b61f04314baccb7c90a + md5: e7a91f0b5a26b7d61ed86423ca9d3d49 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - r-base >=4.4,<4.5.0a0 + - r-lattice >=0.20_27 + license: GPL-2.0-or-later + license_family: GPL3 + size: 1019051 + timestamp: 1757457839052 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/raxml-8.2.13-h7b50bb2_3.tar.bz2 + sha256: 2a974acb73d07dc7e142c997101f14ce865ea4b8f8fb203849002848c37171b9 + md5: 48880314905ab03663a8fb4d97c835ff + depends: + - libgcc >=13 + license: GPL + size: 3227304 + timestamp: 1734060260551 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/raxml-ng-1.2.2-h6747034_2.tar.bz2 + sha256: c59ac13eb664675b08957ced119250e5d9ba08bd4cc3923d84c4fe428d49311b + md5: 4cd32a341c3a8fd75114ba4f2da541b7 + depends: + - gmp >=6.3.0,<7.0a0 + - libgcc >=13 + - libstdcxx >=13 + - openmpi >=4.1.6,<5.0a0 + license: AGPL-3 + license_family: AGPL + size: 1665756 + timestamp: 1744728242938 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/readline-8.2-h8c095d6_2.conda + sha256: 2d6d0c026902561ed77cd646b5021aef2d4db22e57a5b0178dfc669231e06d2c + md5: 283b96675859b20a825f8fa30f311446 + depends: + - libgcc >=13 + - ncurses >=6.5,<7.0a0 + license: GPL-3.0-only + license_family: GPL + size: 282480 + timestamp: 1740379431762 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/reproc-14.2.5.post0-hb9d3cd8_0.conda + sha256: a1973f41a6b956f1305f9aaefdf14b2f35a8c9615cfe5f143f1784ed9aa6bf47 + md5: 69fbc0a9e42eb5fe6733d2d60d818822 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: MIT + license_family: MIT + size: 34194 + timestamp: 1731925834928 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/reproc-cpp-14.2.5.post0-h5888daf_0.conda + sha256: 568485837b905b1ea7bdb6e6496d914b83db57feda57f6050d5a694977478691 + md5: 828302fca535f9cfeb598d5f7c204323 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libstdcxx >=13 + - reproc 14.2.5.post0 hb9d3cd8_0 + license: MIT + license_family: MIT + size: 25665 + timestamp: 1731925852714 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/requests-2.32.5-pyhd8ed1ab_0.conda + sha256: 8dc54e94721e9ab545d7234aa5192b74102263d3e704e6d0c8aa7008f2da2a7b + md5: db0c6b99149880c8ba515cf4abe93ee4 + depends: + - certifi >=2017.4.17 + - charset-normalizer >=2,<4 + - idna >=2.5,<4 + - python >=3.9 + - urllib3 >=1.21.1,<3 + constrains: + - chardet >=3.0.2,<6 + license: Apache-2.0 + license_family: APACHE + size: 59263 + timestamp: 1755614348400 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/rich-14.2.0-pyhcf101f3_0.conda + sha256: edfb44d0b6468a8dfced728534c755101f06f1a9870a7ad329ec51389f16b086 + md5: a247579d8a59931091b16a1e932bbed6 + depends: + - markdown-it-py >=2.2.0 + - pygments >=2.13.0,<3.0.0 + - python >=3.10 + - typing_extensions >=4.0.0,<5.0.0 + - python + license: MIT + license_family: MIT + size: 200840 + timestamp: 1760026188268 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/ruamel.yaml-0.18.16-py311h49ec1c0_0.conda + sha256: 3ad29409ab3742a4cab9e42fc936a2678249c97bbe014905d215afab3aaf13bb + md5: 945689ec8e69e9f254eceef82c9f932f + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + - ruamel.yaml.clib >=0.1.2 + license: MIT + license_family: MIT + size: 276028 + timestamp: 1761160704044 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/ruamel.yaml.clib-0.2.14-py311h49ec1c0_0.conda + sha256: c17b1dc975839a35efc434d3391b8aaa1977a541641d8729c883d5106fb64b97 + md5: 4942e6597e5fc980020b648353c711d4 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: MIT + license_family: MIT + size: 141908 + timestamp: 1760564334661 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/salmon-1.10.3-haf24da9_3.tar.bz2 + sha256: 8b09bc39fd7df1bb325484c69e755b1dac1ef3ddb7ea156dfe5ba60c37949755 + md5: b48fb17ab6ae217737dd84ef4df2f416 + depends: + - boost-cpp + - bzip2 >=1.0.8,<2.0a0 + - icu + - libgcc >=13 + - libjemalloc >=5.3.0 + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + - tbb >=2022.0.0 + license: GPL-3.0-or-later + license_family: GPL3 + size: 6298616 + timestamp: 1733976993298 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/samtools-1.22.1-h96c455f_0.tar.bz2 + sha256: 5850a463a6ccf6a6f94944944d9fa1a49a86942a4365d2aaa035b89ea56ea908 + md5: 7088984cee084e606b2826b1fc3b5906 + depends: + - htslib >=1.22.1,<1.23.0a0 + - libgcc >=13 + - libzlib >=1.3.1,<2.0a0 + - ncurses >=6.5,<7.0a0 + license: MIT + size: 499281 + timestamp: 1752528204243 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/scikit-learn-1.7.2-py311hc3e1efb_0.conda + sha256: c10973e92f71d6a1277a29d3abffefc9ed4b27854b1e3144e505844d7e0a3fe7 + md5: 3f5b4f552d1ef2a5fdc2a4e25db2ee9a + depends: + - __glibc >=2.17,<3.0.a0 + - _openmp_mutex >=4.5 + - joblib >=1.2.0 + - libgcc >=14 + - libstdcxx >=14 + - numpy >=1.22.0 + - numpy >=1.23,<3 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + - scipy >=1.8.0 + - threadpoolctl >=3.1.0 + license: BSD-3-Clause + license_family: BSD + size: 9785405 + timestamp: 1757406401803 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/scipy-1.16.3-py311h1e13796_0.conda + sha256: 3027e8d71a7b7e6b0d14af8f9729ee3923421ff5ee6557f7c7a943786985e524 + md5: 64a45020cd5a51f02fea17ad4dc76535 + depends: + - __glibc >=2.17,<3.0.a0 + - libblas >=3.9.0,<4.0a0 + - libcblas >=3.9.0,<4.0a0 + - libgcc >=14 + - libgfortran + - libgfortran5 >=14.3.0 + - liblapack >=3.9.0,<4.0a0 + - libstdcxx >=14 + - numpy <2.6 + - numpy >=1.23,<3 + - numpy >=1.25.2 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: BSD-3-Clause + license_family: BSD + size: 17213197 + timestamp: 1761691072055 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/sed-4.9-h6688a6e_0.conda + sha256: ee826aa0c6157d4a947722b1205964482ff8e88136bd3161864f8cefdca85b5b + md5: 171afc5f7ca0408bbccbcb69ade85f92 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: GPL-3.0-only + license_family: GPL + size: 228948 + timestamp: 1746562045847 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/sepp-4.5.6-py311haab0aaa_1.conda + sha256: 535733caba2be4e8606189e6b3d1f8d0bedc098d1112f48c4f708437e732ca3c + md5: 974306d6a04301f85cd8832204512be5 + depends: + - dendropy >=5.0.8,<6.0a0 + - hmmer >=3.4,<3.5.0a0 + - libgcc >=13 + - pasta + - pplacer >=1.1.alpha17 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: GPL-3.0-or-later + license_family: GPL3 + size: 1781777 + timestamp: 1755700799279 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/seqkit-2.10.1-he881be0_0.conda + sha256: 1092b10c33272f04a7b9ab074c417d116c80f57ca19ce918969825b8b5205812 + md5: 7aa968b0961e96eb81ad6bfb78711996 + license: MIT + license_family: MIT + size: 12323830 + timestamp: 1755592064787 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/seqmagick-0.8.6-pyhdfd78af_0.tar.bz2 + sha256: 142ff8f89f681c5eeb9429129830d33cf3f8af368109ea9cfb3de5bcced6c3a6 + md5: 48bf0962ccaa5778ed48b50749f4b85d + depends: + - biopython >=1.78 + - pygtrie + - python >3 + license: GNU General Public License (GPL) + license_family: GPL + size: 56328 + timestamp: 1678577572949 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/setuptools-80.9.0-pyhff2d567_0.conda + sha256: 972560fcf9657058e3e1f97186cc94389144b46dbdf58c807ce62e83f977e863 + md5: 4de79c071274a53dcaf2a8c749d1499e + depends: + - python >=3.9 + license: MIT + license_family: MIT + size: 748788 + timestamp: 1748804951958 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/simdjson-4.0.7-hb700be7_0.conda + sha256: 5e29efa1927929885e00909c0386b160d13100a73e031432c42e74df2151f775 + md5: cc9c262a71dd584aa5a3a22fc963255c + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + license: Apache-2.0 + license_family: APACHE + size: 267708 + timestamp: 1759262988515 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/sip-6.10.0-py311h1ddb823_1.conda + sha256: dd954857c0766c2343c416c5087d6bfaca3f4967981339d87fa57b002a77d798 + md5: 8012258dbc1728a96a7a72a2b3daf2ad + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - packaging + - ply + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + - setuptools + - tomli + license: BSD-2-Clause + license_family: BSD + size: 691048 + timestamp: 1759438063197 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/six-1.17.0-pyhe01879c_1.conda + sha256: 458227f759d5e3fcec5d9b7acce54e10c9e1f4f4b7ec978f3bfd54ce4ee9853d + md5: 3339e3b65d58accf4ca4fb8748ab16b3 + depends: + - python >=3.9 + - python + license: MIT + license_family: MIT + size: 18455 + timestamp: 1753199211006 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/sniffio-1.3.1-pyhd8ed1ab_2.conda + sha256: dce518f45e24cd03f401cb0616917773159a210c19d601c5f2d4e0e5879d30ad + md5: 03fe290994c5e4ec17293cfb6bdce520 + depends: + - python >=3.10 + license: Apache-2.0 + license_family: Apache + size: 15698 + timestamp: 1762941572482 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/sqlite-3.51.0-heff268d_0.conda + sha256: 5cece58ca7353705ea47bbe44088baee70d2dfa8bdf2bbcd211698f60ab5e7cd + md5: 5422f0e1b59d2aa29329d5b3e36d57e5 + depends: + - __glibc >=2.17,<3.0.a0 + - icu >=75.1,<76.0a0 + - libgcc >=14 + - libsqlite 3.51.0 hee844dc_0 + - libzlib >=1.3.1,<2.0a0 + - ncurses >=6.5,<7.0a0 + - readline >=8.2,<9.0a0 + license: blessing + size: 182985 + timestamp: 1762299697693 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/sra-tools-3.2.1-h4304569_1.tar.bz2 + sha256: 607b43e8f5b0c9e0e7228bbe19d3d50a6dc54d4b072b0ed508081fe468604264 + md5: e25d29bf7be97231d3d01fedeb7e01c0 + depends: + - ca-certificates + - curl + - libgcc + - libgcc-ng >=12 + - libstdcxx + - libstdcxx-ng >=12 + - ncbi-vdb >=3.2.1 + - ncbi-vdb >=3.2.1,<4.0a0 + - ossuuid + - perl + - perl-uri + - perl-xml-libxml + license: Public Domain + size: 64003013 + timestamp: 1750275321138 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/suitesparse-7.10.1-h5b2951e_7100101.conda + sha256: 7c1c5bd2ba8385202ac3cb4c9c7f9bb87a2e677e977afd72c910314c014b28be + md5: e927e0f248a498050443dab8bb1203a9 + depends: + - libsuitesparseconfig ==7.10.1 h901830b_7100101 + - libamd ==3.3.3 h456b2da_7100101 + - libbtf ==2.3.2 hf02c80a_7100101 + - libcamd ==3.3.3 hf02c80a_7100101 + - libccolamd ==3.3.4 hf02c80a_7100101 + - libcolamd ==3.3.4 hf02c80a_7100101 + - libcholmod ==5.3.1 h9cf07ce_7100101 + - libcxsparse ==4.4.1 hf02c80a_7100101 + - libldl ==3.3.2 hf02c80a_7100101 + - libklu ==2.3.5 h95ff59c_7100101 + - libumfpack ==6.3.5 h873dde6_7100101 + - libparu ==1.0.0 hc6afc67_7100101 + - librbio ==4.3.4 hf02c80a_7100101 + - libspex ==3.2.3 h9226d62_7100101 + - libspqr ==4.3.4 h23b7119_7100101 + license: LGPL-2.1-or-later AND BSD-3-Clause AND GPL-2.0-or-later AND Apache-2.0 + size: 12135 + timestamp: 1741963824816 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/sysroot_linux-64-2.28-h4ee821c_8.conda + sha256: 0053c17ffbd9f8af1a7f864995d70121c292e317804120be4667f37c92805426 + md5: 1bad93f0aa428d618875ef3a588a889e + depends: + - __glibc >=2.28 + - kernel-headers_linux-64 4.18.0 he073ed8_8 + - tzdata + license: LGPL-2.0-or-later AND LGPL-2.0-or-later WITH exceptions AND GPL-2.0-or-later + license_family: GPL + size: 24210909 + timestamp: 1752669140965 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/tar-1.35-h3b78370_0.conda + sha256: eab162cadb948c91c14f18b6f30a7c270d682e820c4f2fdb0030c9fc99b3fa49 + md5: 8e037804a92c84cb901bdd83a8e5ff46 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libiconv >=1.18,<2.0a0 + license: GPL-3.0-or-later + license_family: GPL + size: 778622 + timestamp: 1753467956747 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/tbb-2022.3.0-h8d10470_1.conda + sha256: 2e3238234ae094d5a5f7c559410ea8875351b6bac0d9d0e576bf64b732b8029e + md5: e3259be3341da4bc06c5b7a78c8bf1bd + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libhwloc >=2.12.1,<2.12.2.0a0 + - libstdcxx >=14 + license: Apache-2.0 + size: 181262 + timestamp: 1762509955687 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/tbb-devel-2022.3.0-h74b38a2_1.conda + sha256: 3c1bf7722f5c82459d3580c8e14eed19b08a83188d1c17aad9cb1e34d5f57339 + md5: 11d050030e91674285a5daa2e17b1126 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + - tbb 2022.3.0 h8d10470_1 + size: 1115083 + timestamp: 1762509972811 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/threadpoolctl-3.6.0-pyhecae5ae_0.conda + sha256: 6016672e0e72c4cf23c0cf7b1986283bd86a9c17e8d319212d78d8e9ae42fdfd + md5: 9d64911b31d57ca443e9f1e36b04385f + depends: + - python >=3.9 + license: BSD-3-Clause + license_family: BSD + size: 23869 + timestamp: 1741878358548 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/tk-8.6.13-noxft_ha0e22de_103.conda + sha256: 1544760538a40bcd8ace2b1d8ebe3eb5807ac268641f8acdc18c69c5ebfeaf64 + md5: 86bc20552bf46075e3d92b67f089172d + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libzlib >=1.3.1,<2.0a0 + constrains: + - xorg-libx11 >=1.8.12,<2.0a0 + license: TCL + license_family: BSD + size: 3284905 + timestamp: 1763054914403 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/tktable-2.10-h8d826fa_7.conda + sha256: dd5d8aa7f434acbe4ef5a54f5f2c221650f6c25a4c5ad62c46b36267e1b8fb9b + md5: 3ac51142c19ba95ae0fadefa333c9afb + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - tk >=8.6.13,<8.7.0a0 + license: TCL + size: 92307 + timestamp: 1750266495866 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/toml-0.10.2-pyhd8ed1ab_2.conda + sha256: 5fe40fb250890a1f81be8c5ad0ba94b41ad614ce51e19098110f635dd9400f82 + md5: 00d80af3a7bf27729484e786a68aafff + depends: + - python >=3.10 + license: MIT + license_family: MIT + size: 22702 + timestamp: 1763034696970 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/tomli-2.3.0-pyhcf101f3_0.conda + sha256: cb77c660b646c00a48ef942a9e1721ee46e90230c7c570cdeb5a893b5cce9bff + md5: d2732eb636c264dc9aa4cbee404b1a53 + depends: + - python >=3.10 + - python + license: MIT + license_family: MIT + size: 20973 + timestamp: 1760014679845 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/tomlkit-0.13.3-pyha770c72_0.conda + sha256: f8d3b49c084831a20923f66826f30ecfc55a4cd951e544b7213c692887343222 + md5: 146402bf0f11cbeb8f781fa4309a95d3 + depends: + - python >=3.9 + license: MIT + license_family: MIT + size: 38777 + timestamp: 1749127286558 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/tornado-6.5.2-py311h49ec1c0_2.conda + sha256: 1913516458f92df2a0b415426dce27cc14922415787f4b672a707b233631b1e0 + md5: 8d7a63fc9653ed0bdc253a51d9a5c371 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: Apache-2.0 + license_family: Apache + size: 869926 + timestamp: 1762506861961 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/tqdm-4.67.1-pyhd8ed1ab_1.conda + sha256: 11e2c85468ae9902d24a27137b6b39b4a78099806e551d390e394a8c34b48e40 + md5: 9efbfdc37242619130ea42b1cc4ed861 + depends: + - colorama + - python >=3.9 + license: MPL-2.0 or MIT + size: 89498 + timestamp: 1735661472632 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/transdecoder-5.7.1-pl5321hdfd78af_2.tar.bz2 + sha256: 80f46255e1f872a345f51fd2442a4d981ec6eed484f81e2867e2f9a5548ba101 + md5: f3135d679414069bd2798a5d9ae1faf2 + depends: + - bioconductor-seqlogo + - perl >=5.32.1,<6.0a0 *_perl5 + - perl-db_file + - perl-uri + - r-ggplot2 + license: Broad Institute + size: 16021529 + timestamp: 1752704180167 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/treeshrink-1.3.9-pyhdfd78af_1.tar.bz2 + sha256: 444f80b6c856a7e59b11494295ec79fdbd980bd23ee634247af7952be4b5282a + md5: e1829c427c743378a0ecf0e7b78e8af8 + depends: + - python >=3.6 + - r-base + - r-bms >=0.3.5 + license: GPL-3.0-or-later + license_family: GPL3 + size: 1222053 + timestamp: 1749260070469 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/trimal-1.5.0-h9948957_2.tar.bz2 + sha256: 05fe24e8a05b8303021d6e7451c7c10cfc063942bc96c59d2ed34f343ef0a234 + md5: a4324d1477de960945ab29f954158947 + depends: + - libgcc >=13 + - libstdcxx >=13 + license: GPL-3.0-or-later + license_family: GPL3 + size: 216543 + timestamp: 1733954902060 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/noarch/trimmomatic-0.40-hdfd78af_0.conda + sha256: fa652766f0f73d66b5133df05f6bd3cd1f6f06a39b80bb31f05291ed365d4067 + md5: dcdb3088b9fe9cebff3fd59f2e9ac995 + depends: + - openjdk + - python + license: GPL-3.0-or-later + license_family: GPL3 + size: 197708 + timestamp: 1756377660036 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/trinity-2.15.2-pl5321h077b44d_6.conda + sha256: 165b1040a9a590e1b53d401f656780503a7be87b20dd2b81d3ca7de2de1e0800 + md5: ef395d9988ade2ab9f18def04119ce16 + depends: + - _openmp_mutex >=4.5 + - bioconductor-ctc + - bioconductor-dexseq + - bioconductor-edger + - bioconductor-go.db + - bioconductor-goseq + - bioconductor-qvalue + - bowtie2 >=2.3.0 + - coreutils + - htslib >=1.22.1,<1.23.0a0 + - kallisto + - kmer-jellyfish >=2.3 + - libgcc >=13 + - libgomp + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + - numpy + - openjdk >=17 + - perl >=5.32.1,<5.33.0a0 *_perl5 + - perl-db_file + - python >=3.7 + - r-ape + - r-argparse + - r-base + - r-cluster + - r-fastcluster + - r-gplots + - r-phangorn + - r-sm + - r-tidyverse + - r-vioplot + - salmon + - samtools >=1.14 + - trimmomatic >=0.39 + license: BSD-3-Clause + license_family: BSD + size: 10769197 + timestamp: 1758076050214 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/truststore-0.10.3-pyhe01879c_0.conda + sha256: df334b8978edc4f42e7056764db1a26f1e4c6e6a29d5e2ca426ed5b2f09d24a0 + md5: 15afca3bec34c3ecbeb2028f81a51772 + depends: + - python >=3.10 + - python + license: MIT + license_family: MIT + size: 23801 + timestamp: 1753886790616 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/typing_extensions-4.15.0-pyhcf101f3_0.conda + sha256: 032271135bca55aeb156cee361c81350c6f3fb203f57d024d7e5a1fc9ef18731 + md5: 0caa1af407ecff61170c9437a808404d + depends: + - python >=3.10 + - python + license: PSF-2.0 + license_family: PSF + size: 51692 + timestamp: 1756220668932 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/tzdata-2025b-h78e105d_0.conda + sha256: 5aaa366385d716557e365f0a4e9c3fca43ba196872abbbe3d56bb610d131e192 + md5: 4222072737ccff51314b5ece9c7d6f5a + license: LicenseRef-Public-Domain + size: 122968 + timestamp: 1742727099393 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/ucsc-fatotwobit-482-hdc0a859_0.tar.bz2 + sha256: 49a3812350d8bcde5993e1620e36921e7419fb98db8b160e186fff16a4bfdaf9 + md5: bfac086f1764d2aedfb35540a9009cdf + depends: + - bzip2 >=1.0.8,<2.0a0 + - libgcc >=13 + - liblzma >=5.8.1,<6.0a0 + - libopenssl-static + - libpng >=1.6.49,<1.7.0a0 + - libuuid >=2.38.1,<3.0a0 + - libzlib >=1.3.1,<2.0a0 + - mysql-connector-c >=6.1.11,<6.1.12.0a0 + license: Varies; see https://genome.ucsc.edu/license + size: 383232 + timestamp: 1750330214400 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/linux-64/ucsc-twobitinfo-482-hdc0a859_0.tar.bz2 + sha256: 113321fb70d6cd6b75c04dbd0925cb7a2f4aabfd138d0206af82f899412a1bb6 + md5: 91466585cc720548e4e4c105686fecb8 + depends: + - bzip2 >=1.0.8,<2.0a0 + - libgcc >=13 + - liblzma >=5.8.1,<6.0a0 + - libopenssl-static + - libpng >=1.6.49,<1.7.0a0 + - libuuid >=2.38.1,<3.0a0 + - libzlib >=1.3.1,<2.0a0 + - mysql-connector-c >=6.1.11,<6.1.12.0a0 + license: Varies; see https://genome.ucsc.edu/license + size: 378240 + timestamp: 1750330740699 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/unicodedata2-17.0.0-py311h49ec1c0_1.conda + sha256: d3c0e3ca6eb49095159d8c78970a279a30b98863eff5c3eeb037296d2e1d1670 + md5: 5e6d4026784e83c0a51c86ec428e8cc8 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - python >=3.11,<3.12.0a0 + - python_abi 3.11.* *_cp311 + license: Apache-2.0 + license_family: Apache + size: 408540 + timestamp: 1763054987009 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/urllib3-2.5.0-pyhd8ed1ab_0.conda + sha256: 4fb9789154bd666ca74e428d973df81087a697dbb987775bc3198d2215f240f8 + md5: 436c165519e140cb08d246a4472a9d6a + depends: + - brotli-python >=1.0.9 + - h2 >=4,<5 + - pysocks >=1.5.6,<2.0,!=1.5.7 + - python >=3.9 + - zstandard >=0.18.0 + license: MIT + license_family: MIT + size: 101735 + timestamp: 1750271478254 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/wayland-1.24.0-hd6090a7_1.conda + sha256: 3aa04ae8e9521d9b56b562376d944c3e52b69f9d2a0667f77b8953464822e125 + md5: 035da2e4f5770f036ff704fa17aace24 + depends: + - __glibc >=2.17,<3.0.a0 + - libexpat >=2.7.1,<3.0a0 + - libffi >=3.5.2,<3.6.0a0 + - libgcc >=14 + - libstdcxx >=14 + license: MIT + license_family: MIT + size: 329779 + timestamp: 1761174273487 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/wget-1.21.4-hda4d442_0.conda + sha256: 70df4ac8cca488618458af4705706551cef7e402bac9c2c41dd17148f60cbd1f + md5: 361e96b664eac64a33c20dfd11affbff + depends: + - libgcc-ng >=12 + - libidn2 >=2,<3.0a0 + - libunistring >=0,<1.0a0 + - libzlib >=1.2.13,<2.0.0a0 + - openssl >=3.2.1,<4.0a0 + - zlib + license: GPL-3.0-or-later + license_family: GPL + size: 770380 + timestamp: 1710770110704 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/wheel-0.45.1-pyhd8ed1ab_1.conda + sha256: 1b34021e815ff89a4d902d879c3bd2040bc1bd6169b32e9427497fa05c55f1ce + md5: 75cb7132eb58d97896e173ef12ac9986 + depends: + - python >=3.9 + license: MIT + license_family: MIT + size: 62931 + timestamp: 1733130309598 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xcb-util-0.4.1-h4f16b4b_2.conda + sha256: ad8cab7e07e2af268449c2ce855cbb51f43f4664936eff679b1f3862e6e4b01d + md5: fdc27cb255a7a2cc73b7919a968b48f0 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libxcb >=1.17.0,<2.0a0 + license: MIT + license_family: MIT + size: 20772 + timestamp: 1750436796633 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xcb-util-cursor-0.1.5-hb9d3cd8_0.conda + sha256: c7b35db96f6e32a9e5346f97adc968ef2f33948e3d7084295baebc0e33abdd5b + md5: eb44b3b6deb1cab08d72cb61686fe64c + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libxcb >=1.13 + - libxcb >=1.16,<2.0.0a0 + - xcb-util-image >=0.4.0,<0.5.0a0 + - xcb-util-renderutil >=0.3.10,<0.4.0a0 + license: MIT + license_family: MIT + size: 20296 + timestamp: 1726125844850 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xcb-util-image-0.4.0-hb711507_2.conda + sha256: 94b12ff8b30260d9de4fd7a28cca12e028e572cbc504fd42aa2646ec4a5bded7 + md5: a0901183f08b6c7107aab109733a3c91 + depends: + - libgcc-ng >=12 + - libxcb >=1.16,<2.0.0a0 + - xcb-util >=0.4.1,<0.5.0a0 + license: MIT + license_family: MIT + size: 24551 + timestamp: 1718880534789 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xcb-util-keysyms-0.4.1-hb711507_0.conda + sha256: 546e3ee01e95a4c884b6401284bb22da449a2f4daf508d038fdfa0712fe4cc69 + md5: ad748ccca349aec3e91743e08b5e2b50 + depends: + - libgcc-ng >=12 + - libxcb >=1.16,<2.0.0a0 + license: MIT + license_family: MIT + size: 14314 + timestamp: 1718846569232 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xcb-util-renderutil-0.3.10-hb711507_0.conda + sha256: 2d401dadc43855971ce008344a4b5bd804aca9487d8ebd83328592217daca3df + md5: 0e0cbe0564d03a99afd5fd7b362feecd + depends: + - libgcc-ng >=12 + - libxcb >=1.16,<2.0.0a0 + license: MIT + license_family: MIT + size: 16978 + timestamp: 1718848865819 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xcb-util-wm-0.4.2-hb711507_0.conda + sha256: 31d44f297ad87a1e6510895740325a635dd204556aa7e079194a0034cdd7e66a + md5: 608e0ef8256b81d04456e8d211eee3e8 + depends: + - libgcc-ng >=12 + - libxcb >=1.16,<2.0.0a0 + license: MIT + license_family: MIT + size: 51689 + timestamp: 1718844051451 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xkeyboard-config-2.46-hb03c661_0.conda + sha256: aa03b49f402959751ccc6e21932d69db96a65a67343765672f7862332aa32834 + md5: 71ae752a748962161b4740eaff510258 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - xorg-libx11 >=1.8.12,<2.0a0 + license: MIT + license_family: MIT + size: 396975 + timestamp: 1759543819846 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/xmltodict-1.0.2-pyhcf101f3_0.conda + sha256: 5001a8b0a26810f5b8e20a05bca8cd458b0f6dcfbe00bf648aec017b1fd341f1 + md5: e91301ac8b2ad640df6165db24914332 + depends: + - python >=3.10 + - python + license: MIT + license_family: MIT + size: 20026 + timestamp: 1758191083015 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libice-1.1.2-hb9d3cd8_0.conda + sha256: c12396aabb21244c212e488bbdc4abcdef0b7404b15761d9329f5a4a39113c4b + md5: fb901ff28063514abb6046c9ec2c4a45 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: MIT + license_family: MIT + size: 58628 + timestamp: 1734227592886 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libsm-1.2.6-he73a12e_0.conda + sha256: 277841c43a39f738927145930ff963c5ce4c4dacf66637a3d95d802a64173250 + md5: 1c74ff8c35dcadf952a16f752ca5aa49 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libuuid >=2.38.1,<3.0a0 + - xorg-libice >=1.1.2,<2.0a0 + license: MIT + license_family: MIT + size: 27590 + timestamp: 1741896361728 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libx11-1.8.12-h4f16b4b_0.conda + sha256: 51909270b1a6c5474ed3978628b341b4d4472cd22610e5f22b506855a5e20f67 + md5: db038ce880f100acc74dba10302b5630 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libxcb >=1.17.0,<2.0a0 + license: MIT + license_family: MIT + size: 835896 + timestamp: 1741901112627 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxau-1.0.12-hb03c661_1.conda + sha256: 6bc6ab7a90a5d8ac94c7e300cc10beb0500eeba4b99822768ca2f2ef356f731b + md5: b2895afaf55bf96a8c8282a2e47a5de0 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + license: MIT + license_family: MIT + size: 15321 + timestamp: 1762976464266 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxcomposite-0.4.6-hb9d3cd8_2.conda + sha256: 753f73e990c33366a91fd42cc17a3d19bb9444b9ca5ff983605fa9e953baf57f + md5: d3c295b50f092ab525ffe3c2aa4b7413 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - xorg-libx11 >=1.8.10,<2.0a0 + - xorg-libxfixes >=6.0.1,<7.0a0 + license: MIT + license_family: MIT + size: 13603 + timestamp: 1727884600744 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxcursor-1.2.3-hb9d3cd8_0.conda + sha256: 832f538ade441b1eee863c8c91af9e69b356cd3e9e1350fff4fe36cc573fc91a + md5: 2ccd714aa2242315acaf0a67faea780b + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - xorg-libx11 >=1.8.10,<2.0a0 + - xorg-libxfixes >=6.0.1,<7.0a0 + - xorg-libxrender >=0.9.11,<0.10.0a0 + license: MIT + license_family: MIT + size: 32533 + timestamp: 1730908305254 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxdamage-1.1.6-hb9d3cd8_0.conda + sha256: 43b9772fd6582bf401846642c4635c47a9b0e36ca08116b3ec3df36ab96e0ec0 + md5: b5fcc7172d22516e1f965490e65e33a4 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - xorg-libx11 >=1.8.10,<2.0a0 + - xorg-libxext >=1.3.6,<2.0a0 + - xorg-libxfixes >=6.0.1,<7.0a0 + license: MIT + license_family: MIT + size: 13217 + timestamp: 1727891438799 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxdmcp-1.1.5-hb03c661_1.conda + sha256: 25d255fb2eef929d21ff660a0c687d38a6d2ccfbcbf0cc6aa738b12af6e9d142 + md5: 1dafce8548e38671bea82e3f5c6ce22f + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + license: MIT + license_family: MIT + size: 20591 + timestamp: 1762976546182 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxext-1.3.6-hb9d3cd8_0.conda + sha256: da5dc921c017c05f38a38bd75245017463104457b63a1ce633ed41f214159c14 + md5: febbab7d15033c913d53c7a2c102309d + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - xorg-libx11 >=1.8.10,<2.0a0 + license: MIT + license_family: MIT + size: 50060 + timestamp: 1727752228921 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxfixes-6.0.2-hb03c661_0.conda + sha256: 83c4c99d60b8784a611351220452a0a85b080668188dce5dfa394b723d7b64f4 + md5: ba231da7fccf9ea1e768caf5c7099b84 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - xorg-libx11 >=1.8.12,<2.0a0 + license: MIT + license_family: MIT + size: 20071 + timestamp: 1759282564045 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxi-1.8.2-hb9d3cd8_0.conda + sha256: 1a724b47d98d7880f26da40e45f01728e7638e6ec69f35a3e11f92acd05f9e7a + md5: 17dcc85db3c7886650b8908b183d6876 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - xorg-libx11 >=1.8.10,<2.0a0 + - xorg-libxext >=1.3.6,<2.0a0 + - xorg-libxfixes >=6.0.1,<7.0a0 + license: MIT + license_family: MIT + size: 47179 + timestamp: 1727799254088 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxrandr-1.5.4-hb9d3cd8_0.conda + sha256: ac0f037e0791a620a69980914a77cb6bb40308e26db11698029d6708f5aa8e0d + md5: 2de7f99d6581a4a7adbff607b5c278ca + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - xorg-libx11 >=1.8.10,<2.0a0 + - xorg-libxext >=1.3.6,<2.0a0 + - xorg-libxrender >=0.9.11,<0.10.0a0 + license: MIT + license_family: MIT + size: 29599 + timestamp: 1727794874300 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxrender-0.9.12-hb9d3cd8_0.conda + sha256: 044c7b3153c224c6cedd4484dd91b389d2d7fd9c776ad0f4a34f099b3389f4a1 + md5: 96d57aba173e878a2089d5638016dc5e + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - xorg-libx11 >=1.8.10,<2.0a0 + license: MIT + license_family: MIT + size: 33005 + timestamp: 1734229037766 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxshmfence-1.3.3-hb9d3cd8_0.conda + sha256: c0830fe9fa78d609cd9021f797307e7e0715ef5122be3f784765dad1b4d8a193 + md5: 9a809ce9f65460195777f2f2116bae02 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: MIT + license_family: MIT + size: 12302 + timestamp: 1734168591429 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxt-1.3.1-hb9d3cd8_0.conda + sha256: a8afba4a55b7b530eb5c8ad89737d60d60bc151a03fbef7a2182461256953f0e + md5: 279b0de5f6ba95457190a1c459a64e31 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - xorg-libice >=1.1.1,<2.0a0 + - xorg-libsm >=1.2.4,<2.0a0 + - xorg-libx11 >=1.8.10,<2.0a0 + license: MIT + license_family: MIT + size: 379686 + timestamp: 1731860547604 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxtst-1.2.5-hb9d3cd8_3.conda + sha256: 752fdaac5d58ed863bbf685bb6f98092fe1a488ea8ebb7ed7b606ccfce08637a + md5: 7bbe9a0cc0df0ac5f5a8ad6d6a11af2f + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - xorg-libx11 >=1.8.10,<2.0a0 + - xorg-libxext >=1.3.6,<2.0a0 + - xorg-libxi >=1.7.10,<2.0a0 + license: MIT + license_family: MIT + size: 32808 + timestamp: 1727964811275 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-libxxf86vm-1.1.6-hb9d3cd8_0.conda + sha256: 8a4e2ee642f884e6b78c20c0892b85dd9b2a6e64a6044e903297e616be6ca35b + md5: 5efa5fa6243a622445fdfd72aee15efa + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - xorg-libx11 >=1.8.10,<2.0a0 + - xorg-libxext >=1.3.6,<2.0a0 + license: MIT + license_family: MIT + size: 17819 + timestamp: 1734214575628 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xorg-xextproto-7.3.0-hb9d3cd8_1004.conda + sha256: f302a3f6284ee9ad3b39e45251d7ed15167896564dc33e006077a896fd3458a6 + md5: bc4cd53a083b6720d61a1519a1900878 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: MIT + license_family: MIT + size: 30549 + timestamp: 1726846235301 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/xvfbwrapper-0.2.15-pyhd8ed1ab_0.conda + sha256: a57fc898d14f45865eb15df4500cb805227c5844a17b7eb6b96762059c6ffe08 + md5: 39b56f47421a370e854e525a32cba1f6 + depends: + - python >=3.10 + license: MIT + license_family: MIT + size: 12343 + timestamp: 1761054608318 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xz-5.8.1-hbcc6ac9_2.conda + sha256: 802725371682ea06053971db5b4fb7fbbcaee9cb1804ec688f55e51d74660617 + md5: 68eae977d7d1196d32b636a026dc015d + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - liblzma 5.8.1 hb9d3cd8_2 + - liblzma-devel 5.8.1 hb9d3cd8_2 + - xz-gpl-tools 5.8.1 hbcc6ac9_2 + - xz-tools 5.8.1 hb9d3cd8_2 + license: 0BSD AND LGPL-2.1-or-later AND GPL-2.0-or-later + size: 23987 + timestamp: 1749230104359 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xz-gpl-tools-5.8.1-hbcc6ac9_2.conda + sha256: 840838dca829ec53f1160f3fca6dbfc43f2388b85f15d3e867e69109b168b87b + md5: bf627c16aa26231720af037a2709ab09 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - liblzma 5.8.1 hb9d3cd8_2 + constrains: + - xz 5.8.1.* + license: 0BSD AND LGPL-2.1-or-later AND GPL-2.0-or-later + size: 33911 + timestamp: 1749230090353 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/xz-tools-5.8.1-hb9d3cd8_2.conda + sha256: 58034f3fca491075c14e61568ad8b25de00cb3ae479de3e69be6d7ee5d3ace28 + md5: 1bad2995c8f1c8075c6c331bf96e46fb + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - liblzma 5.8.1 hb9d3cd8_2 + constrains: + - xz 5.8.1.* + license: 0BSD AND LGPL-2.1-or-later + size: 96433 + timestamp: 1749230076687 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/yaml-0.2.5-h280c20c_3.conda + sha256: 6d9ea2f731e284e9316d95fa61869fe7bbba33df7929f82693c121022810f4ad + md5: a77f85f77be52ff59391544bfe73390a + depends: + - libgcc >=14 + - __glibc >=2.17,<3.0.a0 + license: MIT + license_family: MIT + size: 85189 + timestamp: 1753484064210 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/yaml-cpp-0.8.0-h3f2d84a_0.conda + sha256: 4b0b713a4308864a59d5f0b66ac61b7960151c8022511cdc914c0c0458375eca + md5: 92b90f5f7a322e74468bb4909c7354b5 + depends: + - libstdcxx >=13 + - libgcc >=13 + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + license: MIT + license_family: MIT + size: 223526 + timestamp: 1745307989800 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/noarch/yq-3.4.3-pyhe01879c_2.conda + sha256: 9ce1018322a813d27d26f3e834d99c5b0098f55755249478c27cf575d1ecbf91 + md5: 18cefe7c50c1228da474ea0e95a8e646 + depends: + - tomlkit >=0.11.6 + - argcomplete >=1.8.1 + - jq + - python >=3.9 + - pyyaml >=5.3.1 + - setuptools + - toml >=0.10.0 + - xmltodict >=0.11.0 + - python + license: Apache-2.0 + license_family: APACHE + size: 29196 + timestamp: 1749219684767 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/zlib-1.3.1-hb9d3cd8_2.conda + sha256: 5d7c0e5f0005f74112a34a7425179f4eb6e73c92f5d109e6af4ddeca407c92ab + md5: c9f075ab2f33b3bbee9e62d4ad0a6cd8 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libzlib 1.3.1 hb9d3cd8_2 + license: Zlib + license_family: Other + size: 92286 + timestamp: 1727963153079 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/zlib-ng-2.2.5-hde8ca8f_0.conda + sha256: 3a8e7798deafd0722b6b5da50c36b7f361a80b30165d600f7760d569a162ff95 + md5: 1920c3502e7f6688d650ab81cd3775fd + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=14 + - libstdcxx >=14 + license: Zlib + license_family: Other + size: 110843 + timestamp: 1754587144298 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/zstandard-0.25.0-py311haee01d2_1.conda + sha256: d534a6518c2d8eccfa6579d75f665261484f0f2f7377b50402446a9433d46234 + md5: ca45bfd4871af957aaa5035593d5efd2 + depends: + - python + - cffi >=1.11 + - zstd >=1.5.7,<1.5.8.0a0 + - libgcc >=14 + - __glibc >=2.17,<3.0.a0 + - zstd >=1.5.7,<1.6.0a0 + - python_abi 3.11.* *_cp311 + license: BSD-3-Clause + size: 466893 + timestamp: 1762512695614 +- conda: https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/linux-64/zstd-1.5.7-hb8e6e7a_2.conda + sha256: a4166e3d8ff4e35932510aaff7aa90772f84b4d07e9f6f83c614cba7ceefe0eb + md5: 6432cb5d4ac0046c3ac0a8a0f95842f9 + depends: + - __glibc >=2.17,<3.0.a0 + - libgcc >=13 + - libstdcxx >=13 + - libzlib >=1.3.1,<2.0a0 + license: BSD-3-Clause + license_family: BSD + size: 567578 + timestamp: 1742433379869 diff --git a/pixi.toml b/pixi.toml new file mode 100644 index 0000000..2da282b --- /dev/null +++ b/pixi.toml @@ -0,0 +1,63 @@ +[workspace] +authors = ["IvisTang "] +channels = [ + "https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/conda-forge/", + "https://mirrors.tuna.tsinghua.edu.cn/anaconda/cloud/bioconda/", + "conda-forge", + "bioconda" +] +name = "biyelunwen" +platforms = ["linux-64"] +version = "0.1.0" + +[pypi-options] +index-url = "https://mirrors.bfsu.edu.cn/pypi/web/simple" + +[tasks] + +[dependencies] + +[feature.base.dependencies] +sra-tools = "*" +fastp = "*" +trinity = "*" +hisat2 = ">=2.2.1,<3" +samtools = ">=1.22.1,<2" +busco = ">=6.0.0,<7" +transdecoder = ">=5.7.1,<6" +cd-hit = ">=4.8.1,<5" +orthofinder = ">=3.1.0,<4" +pip = ">=25.2,<26" +seqkit = ">=2.10.1,<3" +hmmer = ">=3.4,<4" +r-rstan = ">=2.32.7,<3" +biopython = ">=1.85,<2" +pal2nal = ">=14.1,<15" +trimal = ">=1.5.0,<2" +treeshrink = ">=1.3.9,<2" +r-ape = ">=5.8_1,<6" +raxml-ng = ">=1.2.2,<2" +modeltest-ng = ">=0.1.7,<0.2" +beast = ">=10.5.0,<11" +aster = ">=1.23,<2" +julia = "1.12.*" +conda = ">=25.9.1,<26" +r-phangorn = ">=2.12.1,<3" +r-languageserver = ">=0.3.16,<0.4" +seqmagick = ">=0.8.6,<0.9" +beagle-lib = "4.*" +maven = ">=3.9.11,<4" +matplotlib = ">=3.10.8,<4" +ete3 = ">=3.1.3,<4" +xvfbwrapper = ">=0.2.15,<0.3" +r-ggplot2 = ">=4.0.1,<5" +macse = ">=2.7,<3" + +[feature.mrbayes.dependencies] +beagle-lib = ">=3.1.2,<4" +mrbayes = ">=3.2.7" + +[environments] +default = ["base"] +mrbayes = ["mrbayes"] +