Not a member of Pastebin yet?
Sign Up,
it unlocks many cool features!
- import time
- import math
- import csv
- import os
- class DNA():
- '''This is a class for DNA sequences and their properties. All that you needed is pandas and an excel or csv sheet
- formatted such that the first column is the sequence name and the second column is the sequence.
- Ex:
- Sequence 1, ATGCGCGATCGATCGATGGACTTAGCAAATCGCT
- Sequence 2, ACTACTACTCGCATGCATACGGACGACTGACTG
- ...
- Calculations are taken from http://biotools.nubic.northwestern.edu/OligoCalc.html#helpbasic'''
- def __init__(self,name,sequence):
- # Basic sequence information
- self.name = name
- self.sequence = sequence
- self.baselength = len(sequence)
- # Count the number of the respective nucleotides
- self.num_As = sequence.count('A')
- self.num_Ts = sequence.count('T')
- self.num_Gs = sequence.count('G')
- self.num_Cs = sequence.count('C')
- # Basic Salt information units = M
- self.sodium = 0.050
- self.magnesium = 0
- def GC_content(self):
- '''Simple percentage of G and C out of total sequence'''
- gc = ((self.num_Cs + self.num_Gs) / self.baselength)*100
- return int(gc)
- def AnHydro_MolWeight(self):
- '''Calculation: (Anhydrous Molecular Weight = (An x 313.21) + (Tn x 304.2) + (Cn x 289.18) + (Gn x 329.21) - 61.96)
- Taken from http://biotools.nubic.northwestern.edu/OligoCalc.html#helpbasic'''
- mol = (self.num_As * 313.21) + (self.num_Ts * 304.2) + (self.num_Cs * 289.18) + (self.num_Gs * 329.21) - 61.96
- return int(mol)
- def Melting_Temp(self):
- '''Calculation: (Tm= 100.5 + (41 * (yG+zC)/(wA+xT+yG+zC)) - (820/(wA+xT+yG+zC)) + 16.6*log10([Na+]))
- Taken from http://biotools.nubic.northwestern.edu/OligoCalc.html#helpbasic'''
- tm = Tm= 100.5 + (41 * (self.num_Gs+self.num_Cs)/(self.num_As+self.num_Ts+self.num_Gs+self.num_Cs)) - (820/(self.num_As+self.num_Ts+self.num_Gs+self.num_Cs)) + 16.6*math.log10(self.sodium)
- return int(tm)
- def Estimate_Ext_Coeff(self):
- '''Calculation: (E260 = ( (nA × 15.4) + (nC × 7.4) + (nG × 11.5) + (nT × 8.7) ) × 0.9 × 1000)
- Taken from https://www.atdbio.com/content/1/Ultraviolet-absorbance-of-oligonucleotides#Calculation-of-extinction-coefficient'''
- e = ((self.num_As * 15.4) + (self.num_Cs * 7.4) + (self.num_Gs * 11.5) + (self.num_Ts * 8.7)) * 0.9 * 1000
- return int(e)
- class Optical():
- '''Next is a class made for calculating concentration from absorbance'''
- def __init__(self,absorbance, extinction):
- self.Abs = absorbance
- self.path = 1 # path is either 1 or 0.3 cm from mm
- self.ext_coeff = extinction
- self.dilution = 100
- def Concentration(self):
- conc = ((((self.Abs)/(self.path*self.ext_coeff))*self.dilution)/1E-6)
- return conc
- # name the infile
- print('Please make sure the sequence file is in the (' + os.getcwd() + ') directory')
- file = 'dna_seq.csv' #input('Please enter sequence file name:')
- salt = 0.05 #input('Please enter sodium concentration in M units (so 50 mM would be entered as 0.050):')
- start = time.time()
- # name the outfile
- export_file = 'data.csv'
- # Create csv reader and writers
- with open(export_file, 'w') as data_file:
- with open(file, 'r') as csv_file:
- reader = csv.reader(csv_file)
- header = ['Name',
- 'Sequence',
- 'Number As',
- 'Number Ts',
- 'Number Gs',
- 'Number Cs',
- 'Length',
- 'GC Content (%)',
- 'Formula Weight (g/mol)',
- 'Melting Temp Na Adjusted (C)',
- 'Estimate Extinction (M-1 cm-1)'
- ]
- writer = csv.DictWriter(data_file, fieldnames=header)
- writer.writeheader()
- next(reader)
- sequence_counter = 0
- # call DNA class and define salt based on input
- for rows in reader:
- sequence_counter += 1
- dna = DNA(rows[0], rows[1])
- dna.sodium = float(salt)
- # Create key:value pairs for the different properties calculated
- data = {header[0]: dna.name,
- header[1]: dna.sequence,
- header[2]: dna.num_As,
- header[3]: dna.num_Ts,
- header[4]: dna.num_Gs,
- header[5]: dna.num_Cs,
- header[6]: dna.baselength,
- header[7]: dna.GC_content(),
- header[8]: dna.AnHydro_MolWeight(),
- header[9]: dna.Melting_Temp(),
- header[10]: dna.Estimate_Ext_Coeff()
- }
- writer.writerow(data)
- # Print the file name and the PATH
- print('--Results saved as:(' + export_file + ') --In path:(' + os.getcwd() + ')')
- print('*Beware any files with the name ' + export_file + ' will be overwritten*')
- seconds = time.time() - start
- print('Code took %f seconds to run' % (seconds))
- print('%i sequences were processed'%(sequence_counter))
- input('Hit enter to exit.')
Advertisement
Add Comment
Please, Sign In to add comment