Generalize REMc analysis

2024-07-28 05:19:50 -04:00
parent c8eba4efd4
commit 8de5005055
8 changed files with 179 additions and 11994 deletions
--- a/workflow/templates/qhtcp/REMc/GTF/Concatenate_GTF_results.py
+++ b/workflow/templates/qhtcp/REMc/GTF/Concatenate_GTF_results.py
@@ -1,71 +0,0 @@
-#!/usr/bin/env python
-
-# This code is to concatenate the batch GO Term Finder results (.tsv) generated from batch GTF perl code(Chris Johnson, U of Tulsa) into a list table
-
-import sys, os, string, glob
-
-
-# return the file list
-def list_all_files(Path):
-    list_all_files = []
-    list_all_files = glob.glob(Path +'/*.txt.tsv')
-    return list_all_files
-
-
-# Main function
-
-try:
-   data_file_Path = sys.argv[1]
-   output_file_Path = sys.argv[2]
-except: 
-   print ('Usage: python Concatenate_GTF_results.py /datasetPath /outputFilePath_and_Name')
-   print ('Data file not found, error in given directory')
-   sys.exit(1)
-
-# Open the output file
-
-try:
-    output = open(output_file_Path, 'w')
-except OSError:
-    print ('output file error')
-
-# get all the GTF result files in given directory
-File_list = []
-File_list = list_all_files(data_file_Path)
-File_list.sort()
-
-i = 0
-for file in File_list:
-    #parse the file names given in absolute path
-    file_name = file.strip().split('/')[-1]
-    file_name = file_name.rstrip('.txt.tsv')
-    # function to read tsv files from a given directory
-    #open the file
-    data = open(file,'r')
-    #reading the label line
-    labelLine = data.readline()
-    label = labelLine.strip().split('\t')
-    #write the label
-    #updates2010July26: update following label writing code	
-    if i == 0:
-    #    output.write('cluster origin')
-        for element in label:
-            output.write(element)
-            output.write('\t')
-    i = i + 1
-    #updates2010July26 End
-    #switch to the next line
-    output.write('\n')
-
-    #read the GO terms
-    GOTermLines = data.readlines()
-    for GOTerm in GOTermLines:
-        GOTerm = GOTerm.strip().strip('\t')
-        if GOTerm != '':
-	    #updates2010July26: remove the code to write the first column 'REMc cluster ID'
-            #output.write(file_name)
-            #output.write('\t')
-	    ##updates2010July26 update end
-            output.write(GOTerm + '\n')
-            #output.write('\n')
-output.close()
--- a/workflow/templates/qhtcp/REMc/GTF/terms2tsv_v4.pl
+++ b/workflow/templates/qhtcp/REMc/GTF/terms2tsv_v4.pl
@@ -1,55 +0,0 @@
-#!/usr/bin/perl
-use strict;
-use warnings;
-use diagnostics;
-
-use File::Map qw(map_file);
-
-my $infile = shift;
-
-my $input;
-
-map_file $input, $infile;
-
-{
-	local $_	= $input;
-	(my $f		= $infile) =~ s/(.*\/)?(.*)(\.[^\.]*){2}/$2/;
-	my %orfgene	= (/(Y\w+)\s+(\w+)\n/g);
-	my @indices	= (/\Q-- \E(\d+) of \d+\Q --\E/g);
-	my @ids		= (/GOID\s+GO:(\d+)/g);
-	my @terms	= (/TERM\s+(.*?)\n/g);
-	my @pvalues	= (/\nCORRECTED P-VALUE\s+(\d.*?)\n/g);
-	my @clusterf	= (/NUM_ANNOTATIONS\s+(\d+ of \d+)/g);
-	my @bgfreq	= (/, vs (\d+ of \d+) in the genome/g);
-	my @orfs	= (/The genes annotated to this node are:\n(.*?)\n/g);
-	
-	s/, /:/g for @orfs;
-
-	my @genes;
-	for my $orf (@orfs) {
-		my @otmp = split /:/, $orf;
-		my @gtmp = map { $orfgene{$_} } @otmp;
-		push @genes, (join ':', @gtmp);
-	}
-
-	&header();
-	for my $i (0 .. (@ids - 1)) {
-		&report($f, $ids[$i], $terms[$i], $pvalues[$i], $clusterf[$i], $bgfreq[$i], $orfs[$i], $genes[$i]);
-	}
-}
-
-sub header {
-	print "REMc ID\tGO_term ID\tGO-term\tCluster frequency\tBackground frequency\tP-value\tORFs\tGenes\n";
-}
-
-sub report {
-	my ($f, $id, $term, $p, $cfreq, $bgfreq, $orfs, $genes) = @_;
-
-	$cfreq =~ /(\d+) of (\d+)/;
-	$cfreq = sprintf "%d out of %d genes, %.1f%%", $1, $2, (100*$1/$2);
-
-	$bgfreq =~ /(\d+) of (\d+)/;
-	$bgfreq = sprintf "%d out of %d genes, %.1f%%", $1, $2, (100*$1/$2);
-
-	print "$f\t$id\t$term\t$cfreq\t$bgfreq\t$p\t$orfs\t$genes\n";
-}