annotate vcf_snp.py @ 4:866cd9ce1cbd draft

update anaconda python
author brigidar
date Mon, 02 Nov 2015 19:42:35 -0500
parents ea2f686dfd4a
children 3a352cb57117
Ignore whitespace changes - Everywhere: Within whitespace: At end of lines:
rev   line source
0
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
1 #!/usr/bin/env python
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
2
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
3 #########################################################################################
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
4 # #
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
5 # Name : vcf_snp.py #
2
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
6 # Version : 0.2 #
0
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
7 # Project : extract snp from vcf #
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
8 # Description : Script to exctract snps #
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
9 # Author : Brigida Rusconi #
2
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
10 # Date : November 02 2015 #
0
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
11 # #
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
12 #########################################################################################
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
13
2
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
14 # If the position is identical to the reference it does not print the nucleotide. I have to retrieve it from the ref column.
0
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
15
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
16
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
17 #------------------------------------------------------------------------------------------
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
18
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
19
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
20 import argparse, os, sys, csv, IPython
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
21 import pandas
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
22 import pdb
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
23 from pandas import *
4
866cd9ce1cbd update anaconda python
brigidar
parents: 2
diff changeset
24
0
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
25 #------------------------------------------------------------------------------------------
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
26
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
27
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
28 #output and input file name to give with the script
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
29 parser = argparse.ArgumentParser()
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
30
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
31 parser.add_argument('-o', '--output', help="snp tab")
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
32 parser.add_argument('-s', '--snp_table', help="vcf")
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
33
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
34
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
35 args = parser.parse_args()
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
36 output_file = args.output
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
37 input_file = args.snp_table
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
38 #------------------------------------------------------------------------------------------
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
39
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
40
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
41 #read in file as dataframe
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
42 df =read_csv(input_file,sep='\t', dtype=object)
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
43
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
44 #------------------------------------------------------------------------------------------
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
45
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
46 # only columns with qbase and refbase in table
2
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
47 #count_qbase=list(df.columns.values)
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
48 #qindexes=[]
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
49 #for i, v in enumerate(count_qbase):
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
50 # if 'ALT' in v:
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
51 # qindexes.append(i)
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
52 df2=df.iloc[:,3:5]
0
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
53 #pdb.set_trace()
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
54
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
55 #------------------------------------------------------------------------------------------
2
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
56 ref_list=[]
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
57 for i in range(0,df2.index.size):
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
58 if df2.iloc[i,1]==".":
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
59 ref_list.append(df2.iloc[i,0][0])
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
60 else:
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
61 ref_list.append(df2.iloc[i,1][0])
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
62 #pdb.set_trace()
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
63 #
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
64 ##------------------------------------------------------------------------------------------
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
65 #
0
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
66 #save file with output name for fasta -o option and removes header and index
75cedeb179aa Uploaded
brigidar
parents:
diff changeset
67 with open(output_file,'w') as output:
2
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
68 output.write(df.columns.values[4] + '\t' + ''.join([str(i) for v,i in enumerate(ref_list)]))
ea2f686dfd4a Uploaded
brigidar
parents: 0
diff changeset
69 ##------------------------------------------------------------------------------------------