diff SMART/Java/Python/cleanGff.py @ 46:169d364ddd91

Uploaded
author m-zytnicki
date Mon, 30 Sep 2013 03:19:26 -0400
parents cd852f3e04ab
children
line wrap: on
line diff
--- a/SMART/Java/Python/cleanGff.py	Wed Sep 18 08:51:22 2013 -0400
+++ b/SMART/Java/Python/cleanGff.py	Mon Sep 30 03:19:26 2013 -0400
@@ -43,158 +43,153 @@
 count = {}
 
 class ParsedLine(object):
-	def __init__(self, line, cpt):
-		self.line = line
-		self.cpt  = cpt
-		self.parse()
+    def __init__(self, line, cpt):
+        self.line = line
+        self.cpt  = cpt
+        self.parse()
 
-	def parse(self):
-		self.line = self.line.strip()
-		self.splittedLine = self.line.split(None, 8)
-		if len(self.splittedLine) < 9:
-			raise Exception("Line '%s' has less than 9 fields.  Exiting..." % (self.line))
-		self.type = self.splittedLine[2]
-		self.parseOptions()
-		self.getId()
-		self.getParents()
+    def parse(self):
+        self.line = self.line.strip()
+        self.splittedLine = self.line.split(None, 8)
+        if len(self.splittedLine) < 9:
+            raise Exception("Line '%s' has less than 9 fields.  Exiting..." % (self.line))
+        self.type = self.splittedLine[2]
+        self.parseOptions()
+        self.getId()
+        self.getParents()
 
-	def parseOptions(self):
-		self.parsedOptions = {}
-		for option in self.splittedLine[8].split(";"):
-			option = option.strip()
-			if option == "": continue
-			posSpace = option.find(" ")
-			posEqual = option.find("=")
-			if posEqual != -1 and (posEqual < posSpace or posSpace == -1):
-				key, value = option.split("=", 1)
-			elif posSpace != -1:
-				key, value = option.split(None, 1)
-			else:
-				key   = "ID"
-				value = option
-			self.parsedOptions[key.strip()] = value.strip(" \"")
+    def parseOptions(self):
+        self.parsedOptions = {}
+        for option in self.splittedLine[8].split(";"):
+            option = option.strip()
+            if option == "": continue
+            posSpace = option.find(" ")
+            posEqual = option.find("=")
+            if posEqual != -1 and (posEqual < posSpace or posSpace == -1):
+                key, value = option.split("=", 1)
+            elif posSpace != -1:
+                key, value = option.split(None, 1)
+            else:
+                key   = "ID"
+                value = option
+            self.parsedOptions[key.strip()] = value.strip(" \"")
 
-	def getId(self):
-		for key in self.parsedOptions:
-			if key.lower() == "id":
-				self.id = self.parsedOptions[key]
-				return
-		if "Parent" in self.parsedOptions:
-			parent = self.parsedOptions["Parent"].split(",")[0]
-			if parent not in count:
-				count[parent] = {}
-			if self.type not in count[parent]:
-				count[parent][self.type] = 0
-			count[parent][self.type] += 1
-			self.id = "%s-%s-%d" % (parent, self.type, count[parent][self.type])
-		else:
-			self.id = "smart%d" % (self.cpt)
-		self.parsedOptions["ID"] = self.id
+    def getId(self):
+        for key in self.parsedOptions:
+            if key.lower() == "id":
+                self.id = self.parsedOptions[key]
+                return
+        if "Parent" in self.parsedOptions:
+            parent = self.parsedOptions["Parent"].split(",")[0]
+            if parent not in count:
+                count[parent] = {}
+            if self.type not in count[parent]:
+                count[parent][self.type] = 0
+            count[parent][self.type] += 1
+            self.id = "%s-%s-%d" % (parent, self.type, count[parent][self.type])
+        else:
+            self.id = "smart%d" % (self.cpt)
+        self.parsedOptions["ID"] = self.id
 
-	def getParents(self):
-		for key in self.parsedOptions:
-			if key.lower() in ("parent", "derives_from"):
-				self.parents = self.parsedOptions[key].split(",")
-				return
-		self.parents = None
+    def getParents(self):
+        for key in self.parsedOptions:
+            if key.lower() in ("parent", "derives_from"):
+                self.parents = self.parsedOptions[key].split(",")
+                return
+        self.parents = None
 
-	def removeParent(self):
-		for key in self.parsedOptions.keys():
-			if key.lower() in ("parent", "derives_from"):
-				del self.parsedOptions[key]
+    def removeParent(self):
+        for key in self.parsedOptions.keys():
+            if key.lower() in ("parent", "derives_from"):
+                del self.parsedOptions[key]
 
-	def export(self):
-		self.splittedLine[8] = ";".join(["%s=%s" % (key, value) for key, value in self.parsedOptions.iteritems()])
-		return "%s\n" % ("\t".join(self.splittedLine))
+    def export(self):
+        self.splittedLine[8] = ";".join(["%s=%s" % (key, value) for key, value in self.parsedOptions.iteritems()])
+        return "%s\n" % ("\t".join(self.splittedLine))
 
 
 class CleanGff(object):
 
-	def __init__(self, verbosity = 1):
-		self.verbosity = verbosity
-		self.lines         = {}
-		self.acceptedTypes = []
-		self.parents       = []
-		self.children      = {}
+    def __init__(self, verbosity = 1):
+        self.verbosity = verbosity
+        self.lines         = {}
+        self.acceptedTypes = []
+        self.parents       = []
+        self.children      = {}
 
-	def setInputFileName(self, name):
-		self.inputFile = open(name)
-		
-	def setOutputFileName(self, name):
-		self.outputFile = open(name, "w")
-
-	def setAcceptedTypes(self, types):
-		self.acceptedTypes = types
+    def setInputFileName(self, name):
+        self.inputFile = open(name)
+        
+    def setOutputFileName(self, name):
+        self.outputFile = open(name, "w")
 
-	def parse(self):
-		progress = UnlimitedProgress(100000, "Reading input file", self.verbosity)
-		for cpt, line in enumerate(self.inputFile):
-			if not line or line[0] == "#": continue
-			if line[0] == ">": break
-			parsedLine = ParsedLine(line, cpt)
-			if parsedLine.type in self.acceptedTypes:
-				if parsedLine.id in self.lines:
-					cpt = 1
-					while "%s-%d" % (parsedLine.id, cpt) in self.lines:
-						cpt += 1
-					parsedLine.id = "%s-%d" % (parsedLine.id, cpt)
-				self.lines[parsedLine.id] = parsedLine
-			progress.inc()
-		progress.done()
+    def setAcceptedTypes(self, types):
+        self.acceptedTypes = types
+
+    def parse(self):
+        progress = UnlimitedProgress(100000, "Reading input file", self.verbosity)
+        for cpt, line in enumerate(self.inputFile):
+            if not line or line[0] == "#": continue
+            if line[0] == ">": break
+            parsedLine = ParsedLine(line, cpt)
+            if parsedLine.type in self.acceptedTypes:
+                self.lines[parsedLine.id] = parsedLine
+            progress.inc()
+        progress.done()
 
-	def sort(self):
-		progress = Progress(len(self.lines.keys()), "Sorting file", self.verbosity)
-		for line in self.lines.values():
-			parentFound = False
-			if line.parents:
-				for parent in line.parents:
-					if parent in self.lines:
-						parentFound = True
-						if parent in self.children:
-							self.children[parent].append(line)
-						else:
-							self.children[parent] = [line]
-			if not parentFound:
-				line.removeParent()
-				self.parents.append(line)
-			progress.inc()
-		progress.done()
+    def sort(self):
+        progress = Progress(len(self.lines.keys()), "Sorting file", self.verbosity)
+        for line in self.lines.values():
+            parentFound = False
+            if line.parents:
+                for parent in line.parents:
+                    if parent in self.lines:
+                        parentFound = True
+                        if parent in self.children:
+                            self.children[parent].append(line)
+                        else:
+                            self.children[parent] = [line]
+            if not parentFound:
+                line.removeParent()
+                self.parents.append(line)
+            progress.inc()
+        progress.done()
 
-	def write(self):
-		progress = Progress(len(self.parents), "Writing output file", self.verbosity)
-		for line in self.parents:
-			self.writeLine(line)
-			progress.inc()
-		self.outputFile.close()
-		progress.done()
+    def write(self):
+        progress = Progress(len(self.parents), "Writing output file", self.verbosity)
+        for line in self.parents:
+            self.writeLine(line)
+            progress.inc()
+        self.outputFile.close()
+        progress.done()
 
-	def writeLine(self, line):
-		self.outputFile.write(line.export())
-		if line.id in self.children:
-			for child in self.children[line.id]:
-				self.writeLine(child)
+    def writeLine(self, line):
+        self.outputFile.write(line.export())
+        if line.id in self.children:
+            for child in self.children[line.id]:
+                self.writeLine(child)
 
-	def run(self):
-		self.parse()
-		self.sort()
-		self.write()
+    def run(self):
+        self.parse()
+        self.sort()
+        self.write()
 
 
 if __name__ == "__main__":
-	
-	# parse command line
-	description = "Clean GFF v1.0.3: Clean a GFF file (as given by NCBI) and outputs a GFF3 file. [Category: Other]"
+    
+    # parse command line
+    description = "Clean GFF v1.0.3: Clean a GFF file (as given by NCBI) and outputs a GFF3 file. [Category: Other]"
 
-	parser = OptionParser(description = description)
-	parser.add_option("-i", "--input",     dest="inputFileName",  action="store",                      type="string", help="input file name [compulsory] [format: file in GFF format]")
-	parser.add_option("-o", "--output",    dest="outputFileName", action="store",                      type="string", help="output file [compulsory] [format: output file in GFF3 format]")
-	parser.add_option("-t", "--types",     dest="types",          action="store", default="mRNA,exon", type="string", help="list of comma-separated types that you want to keep [format: string] [default: mRNA,exon]")
-	parser.add_option("-v", "--verbosity", dest="verbosity",      action="store", default=1,           type="int",    help="trace level [format: int]")
-	(options, args) = parser.parse_args()
+    parser = OptionParser(description = description)
+    parser.add_option("-i", "--input",     dest="inputFileName",  action="store",                      type="string", help="input file name [compulsory] [format: file in GFF format]")
+    parser.add_option("-o", "--output",    dest="outputFileName", action="store",                      type="string", help="output file [compulsory] [format: output file in GFF3 format]")
+    parser.add_option("-t", "--types",     dest="types",          action="store", default="mRNA,exon", type="string", help="list of comma-separated types that you want to keep [format: string] [default: mRNA,exon]")
+    parser.add_option("-v", "--verbosity", dest="verbosity",      action="store", default=1,           type="int",    help="trace level [format: int]")
+    (options, args) = parser.parse_args()
 
-	cleanGff = CleanGff(options.verbosity)
-	cleanGff.setInputFileName(options.inputFileName)
-	cleanGff.setOutputFileName(options.outputFileName)
-	cleanGff.setAcceptedTypes(options.types.split(","))
-	cleanGff.run()
+    cleanGff = CleanGff(options.verbosity)
+    cleanGff.setInputFileName(options.inputFileName)
+    cleanGff.setOutputFileName(options.outputFileName)
+    cleanGff.setAcceptedTypes(options.types.split(","))
+    cleanGff.run()