botechga

DNA _analyzerV2

Jul 23rd, 2020
257
0
Never
Not a member of Pastebin yet? Sign Up, it unlocks many cool features!
Python 5.13 KB | None | 0 0
  1. import time
  2. import math
  3. import csv
  4. import os
  5.  
  6.  
  7. class DNA():
  8.     '''This is a class for DNA sequences and their properties. All that you needed is pandas and an excel or csv sheet
  9.    formatted such that the first column is the sequence name and the second column is the sequence.
  10.  
  11.    Ex:
  12.    Sequence 1, ATGCGCGATCGATCGATGGACTTAGCAAATCGCT
  13.    Sequence 2, ACTACTACTCGCATGCATACGGACGACTGACTG
  14.    ...
  15.  
  16.    Calculations are taken from http://biotools.nubic.northwestern.edu/OligoCalc.html#helpbasic'''
  17.  
  18.     def __init__(self,name,sequence):
  19.         #   Basic sequence information
  20.         self.name = name
  21.         self.sequence = sequence
  22.         self.baselength = len(sequence)
  23.  
  24.         #   Count the number of the respective nucleotides
  25.         self.num_As = sequence.count('A')
  26.         self.num_Ts = sequence.count('T')
  27.         self.num_Gs = sequence.count('G')
  28.         self.num_Cs = sequence.count('C')
  29.  
  30.         #   Basic Salt information units = M
  31.         self.sodium = 0.050
  32.         self.magnesium = 0
  33.  
  34.     def GC_content(self):
  35.         '''Simple percentage of G and C out of total sequence'''
  36.         gc = ((self.num_Cs + self.num_Gs) / self.baselength)*100
  37.         return int(gc)
  38.  
  39.  
  40.     def AnHydro_MolWeight(self):
  41.         '''Calculation: (Anhydrous Molecular Weight = (An x 313.21) + (Tn x 304.2) + (Cn x 289.18) + (Gn x 329.21) - 61.96)
  42.        Taken from http://biotools.nubic.northwestern.edu/OligoCalc.html#helpbasic'''
  43.         mol = (self.num_As * 313.21) + (self.num_Ts * 304.2) + (self.num_Cs * 289.18) + (self.num_Gs * 329.21) - 61.96
  44.         return int(mol)
  45.  
  46.     def Melting_Temp(self):
  47.         '''Calculation: (Tm= 100.5 + (41 * (yG+zC)/(wA+xT+yG+zC)) - (820/(wA+xT+yG+zC)) + 16.6*log10([Na+]))
  48.        Taken from http://biotools.nubic.northwestern.edu/OligoCalc.html#helpbasic'''
  49.         tm = Tm= 100.5 + (41 * (self.num_Gs+self.num_Cs)/(self.num_As+self.num_Ts+self.num_Gs+self.num_Cs)) - (820/(self.num_As+self.num_Ts+self.num_Gs+self.num_Cs)) + 16.6*math.log10(self.sodium)
  50.         return int(tm)
  51.  
  52.     def Estimate_Ext_Coeff(self):
  53.         '''Calculation: (E260 = ( (nA × 15.4) + (nC × 7.4) + (nG × 11.5) + (nT × 8.7) ) × 0.9 × 1000)
  54.        Taken from https://www.atdbio.com/content/1/Ultraviolet-absorbance-of-oligonucleotides#Calculation-of-extinction-coefficient'''
  55.         e = ((self.num_As * 15.4) + (self.num_Cs * 7.4) + (self.num_Gs * 11.5) + (self.num_Ts * 8.7)) * 0.9 * 1000
  56.         return int(e)
  57.  
  58. class Optical():
  59.     '''Next is a class made for calculating concentration from absorbance'''
  60.  
  61.     def __init__(self,absorbance, extinction):
  62.         self.Abs = absorbance
  63.         self.path = 1   #   path is either 1 or 0.3 cm from mm
  64.         self.ext_coeff = extinction
  65.         self.dilution =  100
  66.  
  67.     def Concentration(self):
  68.         conc = ((((self.Abs)/(self.path*self.ext_coeff))*self.dilution)/1E-6)
  69.         return conc
  70.  
  71. #   name the infile
  72. print('Please make sure the sequence file is in the (' + os.getcwd() + ') directory')
  73. file =  'dna_seq.csv' #input('Please enter sequence file name:')
  74. salt = 0.05 #input('Please enter sodium concentration in M units (so 50 mM would be entered as 0.050):')
  75. start = time.time()
  76.  
  77. #   name the outfile
  78. export_file = 'data.csv'
  79.  
  80. #   Create csv reader and writers
  81. with open(export_file, 'w') as data_file:
  82.     with open(file, 'r') as csv_file:
  83.         reader = csv.reader(csv_file)
  84.         header = ['Name',
  85.                   'Sequence',
  86.                   'Number As',
  87.                   'Number Ts',
  88.                   'Number Gs',
  89.                   'Number Cs',
  90.                   'Length',
  91.                   'GC Content (%)',
  92.                   'Formula Weight (g/mol)',
  93.                   'Melting Temp Na Adjusted (C)',
  94.                   'Estimate Extinction (M-1 cm-1)'
  95.                   ]
  96.  
  97.         writer = csv.DictWriter(data_file, fieldnames=header)
  98.         writer.writeheader()
  99.         next(reader)
  100.         sequence_counter = 0
  101.         #   call DNA class and define salt based on input
  102.         for rows in reader:
  103.             sequence_counter += 1
  104.             dna = DNA(rows[0], rows[1])
  105.             dna.sodium = float(salt)
  106.  
  107.             #   Create key:value pairs for the different properties calculated
  108.             data = {header[0]: dna.name,
  109.                     header[1]: dna.sequence,
  110.                     header[2]: dna.num_As,
  111.                     header[3]: dna.num_Ts,
  112.                     header[4]: dna.num_Gs,
  113.                     header[5]: dna.num_Cs,
  114.                     header[6]: dna.baselength,
  115.                     header[7]: dna.GC_content(),
  116.                     header[8]: dna.AnHydro_MolWeight(),
  117.                     header[9]: dna.Melting_Temp(),
  118.                     header[10]: dna.Estimate_Ext_Coeff()
  119.                     }
  120.             writer.writerow(data)
  121.  
  122. #   Print the file name and the PATH
  123. print('--Results saved as:(' + export_file + ') --In path:(' + os.getcwd() + ')')
  124. print('*Beware any files with the name ' + export_file + ' will be overwritten*')
  125. seconds = time.time() - start
  126. print('Code took %f seconds to run' % (seconds))
  127. print('%i sequences were processed'%(sequence_counter))
  128. input('Hit enter to exit.')
Advertisement
Add Comment
Please, Sign In to add comment