# HG changeset patch # User m-zytnicki # Date 1369897414 14400 # Node ID cd852f3e04ab8aaed4539a363cda777186b9bbcb # Parent 1236e5a495950d2f1fb37b7c64ec52f3dfbd23b7 Uploaded diff -r 1236e5a49595 -r cd852f3e04ab SMART/Java/Python/cleanGff.py --- a/SMART/Java/Python/cleanGff.py Mon May 13 10:22:25 2013 -0400 +++ b/SMART/Java/Python/cleanGff.py Thu May 30 03:03:34 2013 -0400 @@ -43,153 +43,158 @@ count = {} class ParsedLine(object): - def __init__(self, line, cpt): - self.line = line - self.cpt = cpt - self.parse() + def __init__(self, line, cpt): + self.line = line + self.cpt = cpt + self.parse() - def parse(self): - self.line = self.line.strip() - self.splittedLine = self.line.split(None, 8) - if len(self.splittedLine) < 9: - raise Exception("Line '%s' has less than 9 fields. Exiting..." % (self.line)) - self.type = self.splittedLine[2] - self.parseOptions() - self.getId() - self.getParents() + def parse(self): + self.line = self.line.strip() + self.splittedLine = self.line.split(None, 8) + if len(self.splittedLine) < 9: + raise Exception("Line '%s' has less than 9 fields. Exiting..." % (self.line)) + self.type = self.splittedLine[2] + self.parseOptions() + self.getId() + self.getParents() - def parseOptions(self): - self.parsedOptions = {} - for option in self.splittedLine[8].split(";"): - option = option.strip() - if option == "": continue - posSpace = option.find(" ") - posEqual = option.find("=") - if posEqual != -1 and (posEqual < posSpace or posSpace == -1): - key, value = option.split("=", 1) - elif posSpace != -1: - key, value = option.split(None, 1) - else: - key = "ID" - value = option - self.parsedOptions[key.strip()] = value.strip(" \"") + def parseOptions(self): + self.parsedOptions = {} + for option in self.splittedLine[8].split(";"): + option = option.strip() + if option == "": continue + posSpace = option.find(" ") + posEqual = option.find("=") + if posEqual != -1 and (posEqual < posSpace or posSpace == -1): + key, value = option.split("=", 1) + elif posSpace != -1: + key, value = option.split(None, 1) + else: + key = "ID" + value = option + self.parsedOptions[key.strip()] = value.strip(" \"") - def getId(self): - for key in self.parsedOptions: - if key.lower() == "id": - self.id = self.parsedOptions[key] - return - if "Parent" in self.parsedOptions: - parent = self.parsedOptions["Parent"].split(",")[0] - if parent not in count: - count[parent] = {} - if self.type not in count[parent]: - count[parent][self.type] = 0 - count[parent][self.type] += 1 - self.id = "%s-%s-%d" % (parent, self.type, count[parent][self.type]) - else: - self.id = "smart%d" % (self.cpt) - self.parsedOptions["ID"] = self.id + def getId(self): + for key in self.parsedOptions: + if key.lower() == "id": + self.id = self.parsedOptions[key] + return + if "Parent" in self.parsedOptions: + parent = self.parsedOptions["Parent"].split(",")[0] + if parent not in count: + count[parent] = {} + if self.type not in count[parent]: + count[parent][self.type] = 0 + count[parent][self.type] += 1 + self.id = "%s-%s-%d" % (parent, self.type, count[parent][self.type]) + else: + self.id = "smart%d" % (self.cpt) + self.parsedOptions["ID"] = self.id - def getParents(self): - for key in self.parsedOptions: - if key.lower() in ("parent", "derives_from"): - self.parents = self.parsedOptions[key].split(",") - return - self.parents = None + def getParents(self): + for key in self.parsedOptions: + if key.lower() in ("parent", "derives_from"): + self.parents = self.parsedOptions[key].split(",") + return + self.parents = None - def removeParent(self): - for key in self.parsedOptions.keys(): - if key.lower() in ("parent", "derives_from"): - del self.parsedOptions[key] + def removeParent(self): + for key in self.parsedOptions.keys(): + if key.lower() in ("parent", "derives_from"): + del self.parsedOptions[key] - def export(self): - self.splittedLine[8] = ";".join(["%s=%s" % (key, value) for key, value in self.parsedOptions.iteritems()]) - return "%s\n" % ("\t".join(self.splittedLine)) + def export(self): + self.splittedLine[8] = ";".join(["%s=%s" % (key, value) for key, value in self.parsedOptions.iteritems()]) + return "%s\n" % ("\t".join(self.splittedLine)) class CleanGff(object): - def __init__(self, verbosity = 1): - self.verbosity = verbosity - self.lines = {} - self.acceptedTypes = [] - self.parents = [] - self.children = {} + def __init__(self, verbosity = 1): + self.verbosity = verbosity + self.lines = {} + self.acceptedTypes = [] + self.parents = [] + self.children = {} - def setInputFileName(self, name): - self.inputFile = open(name) - - def setOutputFileName(self, name): - self.outputFile = open(name, "w") + def setInputFileName(self, name): + self.inputFile = open(name) + + def setOutputFileName(self, name): + self.outputFile = open(name, "w") + + def setAcceptedTypes(self, types): + self.acceptedTypes = types - def setAcceptedTypes(self, types): - self.acceptedTypes = types - - def parse(self): - progress = UnlimitedProgress(100000, "Reading input file", self.verbosity) - for cpt, line in enumerate(self.inputFile): - if not line or line[0] == "#": continue - if line[0] == ">": break - parsedLine = ParsedLine(line, cpt) - if parsedLine.type in self.acceptedTypes: - self.lines[parsedLine.id] = parsedLine - progress.inc() - progress.done() + def parse(self): + progress = UnlimitedProgress(100000, "Reading input file", self.verbosity) + for cpt, line in enumerate(self.inputFile): + if not line or line[0] == "#": continue + if line[0] == ">": break + parsedLine = ParsedLine(line, cpt) + if parsedLine.type in self.acceptedTypes: + if parsedLine.id in self.lines: + cpt = 1 + while "%s-%d" % (parsedLine.id, cpt) in self.lines: + cpt += 1 + parsedLine.id = "%s-%d" % (parsedLine.id, cpt) + self.lines[parsedLine.id] = parsedLine + progress.inc() + progress.done() - def sort(self): - progress = Progress(len(self.lines.keys()), "Sorting file", self.verbosity) - for line in self.lines.values(): - parentFound = False - if line.parents: - for parent in line.parents: - if parent in self.lines: - parentFound = True - if parent in self.children: - self.children[parent].append(line) - else: - self.children[parent] = [line] - if not parentFound: - line.removeParent() - self.parents.append(line) - progress.inc() - progress.done() + def sort(self): + progress = Progress(len(self.lines.keys()), "Sorting file", self.verbosity) + for line in self.lines.values(): + parentFound = False + if line.parents: + for parent in line.parents: + if parent in self.lines: + parentFound = True + if parent in self.children: + self.children[parent].append(line) + else: + self.children[parent] = [line] + if not parentFound: + line.removeParent() + self.parents.append(line) + progress.inc() + progress.done() - def write(self): - progress = Progress(len(self.parents), "Writing output file", self.verbosity) - for line in self.parents: - self.writeLine(line) - progress.inc() - self.outputFile.close() - progress.done() + def write(self): + progress = Progress(len(self.parents), "Writing output file", self.verbosity) + for line in self.parents: + self.writeLine(line) + progress.inc() + self.outputFile.close() + progress.done() - def writeLine(self, line): - self.outputFile.write(line.export()) - if line.id in self.children: - for child in self.children[line.id]: - self.writeLine(child) + def writeLine(self, line): + self.outputFile.write(line.export()) + if line.id in self.children: + for child in self.children[line.id]: + self.writeLine(child) - def run(self): - self.parse() - self.sort() - self.write() + def run(self): + self.parse() + self.sort() + self.write() if __name__ == "__main__": - - # parse command line - description = "Clean GFF v1.0.3: Clean a GFF file (as given by NCBI) and outputs a GFF3 file. [Category: Other]" + + # parse command line + description = "Clean GFF v1.0.3: Clean a GFF file (as given by NCBI) and outputs a GFF3 file. [Category: Other]" - parser = OptionParser(description = description) - parser.add_option("-i", "--input", dest="inputFileName", action="store", type="string", help="input file name [compulsory] [format: file in GFF format]") - parser.add_option("-o", "--output", dest="outputFileName", action="store", type="string", help="output file [compulsory] [format: output file in GFF3 format]") - parser.add_option("-t", "--types", dest="types", action="store", default="mRNA,exon", type="string", help="list of comma-separated types that you want to keep [format: string] [default: mRNA,exon]") - parser.add_option("-v", "--verbosity", dest="verbosity", action="store", default=1, type="int", help="trace level [format: int]") - (options, args) = parser.parse_args() + parser = OptionParser(description = description) + parser.add_option("-i", "--input", dest="inputFileName", action="store", type="string", help="input file name [compulsory] [format: file in GFF format]") + parser.add_option("-o", "--output", dest="outputFileName", action="store", type="string", help="output file [compulsory] [format: output file in GFF3 format]") + parser.add_option("-t", "--types", dest="types", action="store", default="mRNA,exon", type="string", help="list of comma-separated types that you want to keep [format: string] [default: mRNA,exon]") + parser.add_option("-v", "--verbosity", dest="verbosity", action="store", default=1, type="int", help="trace level [format: int]") + (options, args) = parser.parse_args() - cleanGff = CleanGff(options.verbosity) - cleanGff.setInputFileName(options.inputFileName) - cleanGff.setOutputFileName(options.outputFileName) - cleanGff.setAcceptedTypes(options.types.split(",")) - cleanGff.run() + cleanGff = CleanGff(options.verbosity) + cleanGff.setInputFileName(options.inputFileName) + cleanGff.setOutputFileName(options.outputFileName) + cleanGff.setAcceptedTypes(options.types.split(",")) + cleanGff.run()